[{"id":"pi-squared-reasoning-data-2026","title":"$π^2$: Structure-Originated Reasoning Data Improves Long-Context Reasoning Ability of Large Language Models","year":2026,"venue":"arXiv preprint; under review (2026)","authors":["Quyet V. Do","Thinh Pham","Nguyen Nguyen","Sha Li","Pratibha Zunjare","Tu Vu"],"authors_zh":"Quyet V. Do, Thinh Pham, Nguyen Nguyen, Sha Li, Pratibha Zunjare, Tu Vu","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["long-context-reasoning"],"tags":["long-context-data","dual-execution-verification","reasoning-traces"],"status":"verified","priority":"可读","paper_type_zh":"长上下文推理数据集、生成流程与评测集","best_for_zh":"适合构造可验证多文档推理 SFT 数据的研究者。","confidence":"high","one_line":["π2 turns Wikipedia tables into long-context QA with SQL/Python-agreed answers and back-translated reasoning traces for data-efficient SFT.","π2 将 Wikipedia 表格转成由 SQL 与 Python 双路径确认答案的长上下文问答，再反向生成推理轨迹用于高效 SFT。"],"why":"It combines realistic unstructured contexts with an executable ground-truth path and releases the resulting training records.","primary_link":"https://arxiv.org/abs/2604.05114","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/vtpss/pi-squared"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/drive/folders/1ISRgLijDFvwFpHzV4tmHix6MIwIGG2gd?usp=sharing"}],"link_count":3,"sections":9},{"id":"humanitys-last-exam-2025","title":"A benchmark of expert-level academic questions to assess AI capabilities","year":2026,"venue":"Nature 649, 1139-1146 (2026) / arXiv:2501.14249","authors":["Center for AI Safety","Scale AI","HLE Contributors Consortium"],"authors_zh":"Long Phan 等（Center for AI Safety、Scale AI、HLE Contributors Consortium 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["frontier-academic-reasoning","multimodal","science","mathematics","humanities","contamination-audit"],"tags":["benchmark","expert-written","frontier-evaluation","multimodal","contamination-audit"],"status":"verified","priority":"必读","paper_type_zh":"Nature 649, 1139-1146 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注专家 benchmark、短答自动评分、公开题库污染、版本治理和多模态评测资产的读者。","confidence":"high","one_line":["Humanity's Last Exam releases an expert-written frontier academic benchmark with closed-ended multimodal and text questions graded by reference answers.","Humanity's Last Exam 汇集全球专家编写的闭合式学术难题，用多选/短答参考答案评测 frontier model 的知识与推理边界。"],"why":"It exposes the audit tension between public expert benchmarks, short-answer grading, version drift, and future training contamination.","primary_link":"https://arxiv.org/abs/2501.14249","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/centerforaisafety/hle"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/cais/hle"},{"key":"project","label":["Project","项目主页"],"url":"https://lastexam.ai/"}],"link_count":6,"sections":9},{"id":"process-reward-models-survey-2026","title":"A Comprehensive Survey of Process Reward Models: Data Generation, Model Construction, and Usage","year":2026,"venue":"ACL 2026 (Long Papers)","authors":["Congmin Zheng","Jiachen Zhu","Zhuoying Ou","Yuxiang Chen","Kangning Zhang","Rong Shan","Zeyu Zheng","Mengyue Yang","Jianghao Lin","Yong Yu","Weinan Zhang"],"authors_zh":"Congmin Zheng 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["step_level","process_reward","trajectory_value"],"training_use":["process_supervision","reward_modeling","rlvr","test_time_compute"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["process-supervision","reward-modeling","reasoning-data","test-time-scaling"],"tags":["foundations-and-primers","process-reward-models","process-supervision","reasoning-traces","acl-2026"],"status":"verified","priority":"必读","paper_type_zh":"过程奖励模型与过程监督综述","best_for_zh":"希望系统理解过程监督如何从数据走向训练与推理使用的读者。","confidence":"high","one_line":["A 2026 ACL survey that traces process reward models from labeled reasoning data through construction to search and reinforcement-learning use.","贯通过程数据生成、过程奖励模型构建与使用的 2026 年 ACL 综述。"],"why":"It explains why a reasoning trace becomes process supervision only once intermediate states receive an explicit feedback contract.","primary_link":"https://aclanthology.org/2026.acl-long.163/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/despzcm/Survey-of-Process-Reward-Model"}],"link_count":4,"sections":9},{"id":"contaminated-reasoning-geometry-2026","title":"A Narrowing Geometry in Contaminated Reasoning","year":2026,"venue":"ICML 2026","authors":["Jiakuan Xie","Pengfei Cao","Kang Liu","Jun Zhao"],"authors_zh":"Jiakuan Xie, Pengfei Cao, Kang Liu, Jun Zhao","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"污染推理机制分析与干预研究","best_for_zh":"需要诊断泄露样本中的记忆化推理、比较污染模型与基础模型的人。","confidence":"medium","one_line":["A mechanistic account links contaminated reasoning to geometric narrowing and tests an intervention that restores base-model consistency.","从内部几何收缩解释污染推理，并用干预部分恢复与基础模型的一致性。"],"why":"It gives a concrete mechanism and intervention for auditing unreliable reasoning on leaked inputs.","primary_link":"https://openreview.net/forum?id=jXhXr53hdG","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/jiakuan929/ContamReasoning"}],"link_count":2,"sections":9},{"id":"post-training-reasoning-data-primer-2026","title":"A Primer in Post-Training Reasoning Data: What We Know About How It Works","year":2026,"venue":"arXiv preprint","authors":["Yaoming Li","Guangxiang Zhao","Qilong Shi","Lin Sun","Xiangzheng Zhang","Tong Yang"],"authors_zh":"Yaoming Li, Guangxiang Zhao, Qilong Shi, Lin Sun, Xiangzheng Zhang, Tong Yang","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["post-training-reasoning-data"],"tags":["arxiv-2606.02113","foundations_and_primers","post-training-data-primer"],"status":"verified","priority":"可读","paper_type_zh":"后训练推理数据导读","best_for_zh":"适合判断应收集什么推理数据、如何验证，以及哪些规模结论能迁移到自身场景的研究人员。","confidence":"high","one_line":["This primer organizes more than 150 studies into a practical map of reasoning-data objects, utility signals, construction choices, and scaling limits.","这篇 primer 把 150 多项研究整理为推理数据对象、效用信号、构造选择和规模边界的实用地图。"],"why":"It prevents data recipes from being compared without their verifier, base model, trajectory, scale, and lineage context.","primary_link":"https://arxiv.org/abs/2606.02113","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/RenBing-Sumeru/Awesome-LLM-Reasoning-Data"}],"link_count":2,"sections":9},{"id":"inductive-reasoning-survey-2026","title":"A Survey of Inductive Reasoning for Large Language Models","year":2026,"venue":"ACL 2026 (Long Papers)","authors":["Kedi Chen","Dezhao Ruan","Yuhao Dan","Yaoting Wang","Siyu Yan","Xuecheng Wu","Yinqi Zhang","Qin Chen","Jie Zhou","Liang He","Biqing Qi","Linyang Li","Qipeng Guo","Xiaoming Shi","Wei Zhang"],"authors_zh":"Kedi Chen 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["inductive-reasoning","reasoning-data","test-time-scaling","benchmarks"],"tags":["foundations-and-primers","inductive-reasoning","acl-2026","survey"],"status":"verified","priority":"可读","paper_type_zh":"归纳推理综述","best_for_zh":"想理解模型如何从具体观察归纳出一般规则的读者。","confidence":"high","one_line":["An ACL 2026 survey that organizes inductive reasoning methods by post-training, exploration, data augmentation, and evaluation.","系统梳理大语言模型归纳推理的训练、探索、数据增强与评测方法。"],"why":"It distinguishes learning a general rule from merely producing a plausible answer on a familiar pattern.","primary_link":"https://aclanthology.org/2026.acl-long.1447/","links":[],"link_count":2,"sections":9},{"id":"reasoning-intensive-retrieval-survey-2026","title":"A Survey of Reasoning-Intensive Retrieval: Progress and Challenges","year":2026,"venue":"ACL 2026","authors":["Yiyang Wei","Tingyu Song","Siyue Zhang","Yilun Zhao"],"authors_zh":"Yiyang Wei 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["reward_verifier_layer","trace_writing"],"domains":["retrieval","reasoning","evaluation"],"tags":["foundations-and-primers","retrieval","reasoning","acl-2026","survey"],"status":"verified","priority":"可读","paper_type_zh":"推理密集检索综述","best_for_zh":"需要设计或比较推理型检索、重排序与评测的读者。","confidence":"high","one_line":["An ACL 2026 survey of retrieval tasks where finding evidence depends on an inferential connection, not only semantic similarity.","梳理需要依靠隐含推断关系而非仅靠语义相似度的检索任务、基准与方法。"],"why":"It helps readers separate ordinary semantic retrieval from retrieval that needs multi-step or latent-link reasoning.","primary_link":"https://aclanthology.org/2026.acl-long.1949/","links":[],"link_count":2,"sections":9},{"id":"abc-bench-agentic-backend-coding-2026","title":"ABC-Bench: Benchmarking Agentic Backend Coding in Real-World Development","year":2026,"venue":"Findings of ACL 2026","authors":["Jie Yang","Honglin Guo","Li Ji","Jiazheng Zhou","Rui Zheng","Zhikai Lei","Shuo Zhang","Zhiheng Xi","Shichun Liu","Yuxin Wang","Bo Wang","Yining Zheng","Tao Gui","Xipeng Qiu"],"authors_zh":"Jie Yang, Honglin Guo, Li Ji, Jiazheng Zhou, Rui Zheng, Zhikai Lei, Shuo Zhang, Zhiheng Xi, Shichun Liu, Yuxin Wang, Bo Wang, Yining Zheng, Tao Gui, Xipeng Qiu","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","backend-development","agent-evaluation"],"tags":["programmatic-verification","benchmark","2026"],"status":"verified","priority":"必读","paper_type_zh":"容器化后端开发智能体基准","best_for_zh":"需要评测后端服务部署、跨文件修改和端到端 API 验证的代码智能体研究者。","confidence":"high","one_line":["ABC-Bench evaluates agents on 224 containerized backend-development tasks whose deployed services must pass end-to-end HTTP API tests.","ABC-Bench 将 224 个后端开发任务放入容器化服务环境，以部署后的真实 HTTP API 测试验证智能体的端到端实现。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://aclanthology.org/2026.findings-acl.1142/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenMOSS/ABC-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenMOSS-Team/ABC-Bench"}],"link_count":3,"sections":9},{"id":"active-dpo-preference-selection-2026","title":"ActiveDPO: Active Direct Preference Optimization for Sample-Efficient Alignment","year":2026,"venue":"ICLR 2026","authors":["Xiaoqiang Lin","Arun Verma","Zhongxiang Dai","Daniela Rus","See-Kiong Ng","Bryan Kian Hsiang Low"],"authors_zh":"Xiaoqiang Lin、Arun Verma、Zhongxiang Dai、Daniela Rus、See-Kiong Ng、Bryan Kian Hsiang Low（新加坡国立大学、SMART、中山大学深圳校区、麻省理工学院）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","summarization","question-answering"],"tags":["active-learning","direct-preference-optimization","preference-data-selection","gradient-uncertainty"],"status":"verified","priority":"必读","paper_type_zh":"主动偏好数据获取与直接偏好优化研究","best_for_zh":"适合设计标签高效偏好优化、且需要支付昂贵人工判断成本的读者。","confidence":"high","one_line":["ActiveDPO uses the target policy's projected implicit-reward gradients to actively acquire diverse, high-uncertainty preference pairs for iterative DPO.","ActiveDPO 利用目标策略投影后的隐式奖励梯度，主动获取多样且高不确定性的偏好对，再进行迭代式 DPO 训练。"],"why":"It turns the current policy's own gradient geometry into an explicit and theoretically motivated rule for purchasing preference labels.","primary_link":"https://openreview.net/forum?id=RD4XgyVyGh","links":[],"link_count":3,"sections":9},{"id":"adaptive-generate-rank-verify-2026","title":"Adaptive Generate-Rank-Verify: Inference-Time Search with Costly Verification","year":2026,"venue":"arXiv preprint","authors":["Shaddin Dughmi","Mahdi Haghifam","Yusuf Hakan Kalayci"],"authors_zh":"Shaddin Dughmi、Yusuf Hakan Kalayci（南加州大学）；Mahdi Haghifam（芝加哥丰田理工学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code-generation"],"tags":["test-time-compute","verification","adaptive-search","reward-ranking","theory"],"status":"verified","priority":"必读","paper_type_zh":"带昂贵验证器的自适应测试时搜索研究","best_for_zh":"研究验证器调用预算、代码隐藏测试或可判定推理搜索的读者。","confidence":"high","one_line":["ADAP adaptively grows a candidate pool and verifies its top-ranked members, with a constant-factor cost guarantee under a monotone score-success assumption.","ADAP 根据廉价排序分数逐步扩大候选池，并只为高分候选支付昂贵验证成本，在单调假设下接近最优开销。"],"why":"It makes verifier calls a measured budget rather than a free post-processing step and exposes the assumption needed for adaptive ranking to work.","primary_link":"https://arxiv.org/abs/2605.17609","links":[],"link_count":2,"sections":9},{"id":"adaptive-test-time-compute-constrained-policy-2026","title":"Adaptive Test-Time Compute Allocation for Reasoning LLMs via Constrained Policy Optimization","year":2026,"venue":"arXiv preprint","authors":["Zhiyuan Zhai","Bingcong Li","Bingnan Xiao","Ming Li","Xin Wang"],"authors_zh":"Zhiyuan Zhai；Bingcong Li；Bingnan Xiao；Ming Li；Xin Wang","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["search_substrate","optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["adaptive-compute","constrained-optimization","test-time-compute","budget-allocation","rlvr"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A budget-allocation method that turns repeated responses into supervised compute-policy labels.","AdaCompute-LLM 为问题—预算组合收集 48 个响应，以受约束目标的拉格朗日标签训练分类器，从而按问题分配推理计算量。"],"why":"It makes the missing bridge between test-time scaling and reusable trace data clear: question–budget rollouts, label derivation, and costs must all be retained.","primary_link":"https://arxiv.org/abs/2604.14853","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhiyuanZhai20/AdaCompute-LLM"}],"link_count":3,"sections":9},{"id":"categorical-verification-tts-2026","title":"Adaptive Test-Time Compute Allocation via Learned Heuristics over Categorical Structure","year":2026,"venue":"arXiv preprint","authors":["Shuhui Qu"],"authors_zh":"Shuhui Qu（机构：斯坦福大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","verifier-allocation","process-reward-model","search","structured-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"选择性验证与自适应测试时分配研究","best_for_zh":"优化结构化推理搜索中昂贵验证调用的读者。","confidence":"high","one_line":["The method saves verifier calls by gating invalid moves, ranking viable moves, and allocating a local verification budget at each reasoning state.","该方法通过筛除无效动作、排序可行动作并逐状态分配验证预算来节约验证调用。"],"why":"It moves compute allocation from whole-problem sample counts to the ambiguous intermediate decisions where verification has the most marginal value.","primary_link":"https://arxiv.org/abs/2602.03975","links":[],"link_count":2,"sections":9},{"id":"evolving-icl-ttc-allocation-2026","title":"Adaptive Test-Time Compute Allocation with Evolving In-Context Demonstrations","year":2026,"venue":"arXiv preprint","authors":["Bowen Zuo","Dongruo Zhou","Yinglun Zhu"],"authors_zh":"Bowen Zuo、Dongruo Zhou、Yinglun Zhu（机构：加州大学河滨分校、印第安纳大学布卢明顿分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","software-engineering","general-reasoning"],"tags":["test-time-compute","adaptive-allocation","in-context-learning","retrieval","verification"],"status":"verified","priority":"可读","paper_type_zh":"自适应测试时计算分配与演化上下文学习研究","best_for_zh":"研究测试集级计算分配与检索条件化测试时扩展的读者。","confidence":"high","one_line":["The method reallocates later inference rounds to unresolved queries and conditions them on successful, semantically related test-time demonstrations.","该方法把后续推理轮次分配给未解决问题，并用语义相近的成功测试时示例改变其生成条件。"],"why":"It exposes that a test-time budget changes not only how many samples are generated but also which successful generations can alter future sampling.","primary_link":"https://arxiv.org/abs/2604.21018","links":[],"link_count":2,"sections":9},{"id":"sonata-adaptive-thinking-2026","title":"Adaptive Thinking: Large Language Models Know When to Think in Latent Space","year":2026,"venue":"ICLR 2026","authors":["Pingzhi Li","Bairu Hou","Yun Zhu","Yihao Feng","Ke Ye","Tao Lei","Zhifeng Chen","Tianlong Chen","Xianzhi Du"],"authors_zh":"Pingzhi Li、Bairu Hou、Yun Zhu、Yihao Feng、Ke Ye、Tao Lei、Zhifeng Chen、Tianlong Chen、Xianzhi Du（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["Adaptive Thinking: Large Language Models Know When to Think in Latent Space","固定思考预算会在简单提示上浪费 token、在困难提示上投入不足，而查询复杂度与计算最优预算的关系并不清楚。"],"why":"It makes an inference-budget decision auditable.","primary_link":"https://iclr.cc/virtual/2026/poster/10011708","links":[{"key":"project","label":["Project","项目主页"],"url":"https://machinelearning.apple.com/research/adaptive-thinking"}],"link_count":2,"sections":9},{"id":"block-diffusion-tts-2026","title":"Advancing Block Diffusion Language Models for Test-Time Scaling","year":2026,"venue":"arXiv preprint","authors":["Yi Lu","Deyang Kong","Jianing Wang","Linsen Guo","Xue Wang","Qi Guo","Tao Gui","Xuanjing Huang","Wei Ye","Shikun Zhang","Wei Wang"],"authors_zh":"Yi Lu、Tao Gui、Xuanjing Huang（复旦大学）；Deyang Kong、Qi Guo、Wei Ye、Shikun Zhang（北京大学）；Jianing Wang、Linsen Guo、Xue Wang、Wei Wang（美团 LongCat 团队）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","block-diffusion","adaptive-decoding","reasoning","inference-efficiency"],"status":"verified","priority":"可读","paper_type_zh":"块扩散语言模型的测试时扩展研究","best_for_zh":"希望比较非自回归推理扩展机制与采样预算机制的读者。","confidence":"high","one_line":["The paper makes block-diffusion reasoning scale at test time by adapting denoising and using coarse exploration followed by fine-grained critique.","该研究让块扩散推理在测试时同时自适应控制去噪和推理块大小，以平衡探索、细化与速度。"],"why":"It measures an alternative compute unit—block size and denoising work—while exposing the accuracy-speed trade-off on long reasoning.","primary_link":"https://arxiv.org/abs/2602.09555","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LuLuLuyi/TDAR"}],"link_count":3,"sections":9},{"id":"advermcts-2026","title":"AdverMCTS: Combating Pseudo-Correctness in Code Generation via Adversarial Monte Carlo Tree Search","year":2026,"venue":"ICML 2026","authors":["Qingyao Li","Weiwen Liu","Weinan Zhang","Yong Yu","Bo An"],"authors_zh":"Qingyao Li、Weiwen Liu、Weinan Zhang、Yong Yu、Bo An","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","agent_environment","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","frontier_pipeline","scaling_report","release_audit"],"domains":["code","software_engineering"],"tags":["adversarial-mcts","code-generation","counterexample-generation","dynamic-test-memory","output-arbiter","pseudo-correctness","test-time-compute","verifier-refresh","open-code"],"status":"partial","priority":"必读","paper_type_zh":"对抗式测试时搜索、动态验证器与代码选择方法","best_for_zh":"研究代码推理、MCTS、反例生成、动态验证器和验证器审计的读者","confidence":"medium","one_line":["AdverMCTS co-evolves a Solver code-search tree and an Attacker test-search tree, retaining Arbiter-labeled counterexamples as a growing hard verifier at inference time.","AdverMCTS 让 Solver 代码树与 Attacker 测试树共同演化，并把同一主干模型的 Arbiter 标注反例存为逐题动态硬验证器；但 79.88% 测试有效率与 16.1% 误杀率表明，持久约束仍可能被错误标签污染。"],"why":"It turns adversarial counterexample discovery into an explicit inference-time data pipeline, while making clear that test validity, sandboxing, retention policy, decision lineage and release completeness determine whether the resulting verifier can be trusted or reused.","primary_link":"https://openreview.net/forum?id=0JpXqndUrA","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SIMONLQY/AdverMCTS"}],"link_count":4,"sections":9},{"id":"aegis-2026","title":"Aegis: Automated Error Generation and Attribution for Multi-Agent Systems","year":2026,"venue":"ICLR 2026","authors":["Fanqi Kong","Ruijie Zhang","Huaxiao Yin","Guibin Zhang","Xiaofei Zhang","Ziang Chen","Zhaowei Zhang","Xiaoyuan Zhang","Song-Chun Zhu","Xue Feng"],"authors_zh":"Fanqi Kong, Ruijie Zhang, Huaxiao Yin, Guibin Zhang, Xiaofei Zhang, Ziang Chen, Zhaowei Zhang, Xiaoyuan Zhang, Song-Chun Zhu, Xue Feng","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","benchmark","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multi_agent_systems","agent_trajectories","error_attribution","mathematics","code","science","knowledge","general_assistant"],"tags":["multi-agent-systems","agent-trajectories","synthetic-failures","error-attribution","prompt-injection","response-corruption","mixed-verifier","sft","grpo","contrastive-learning","selection-bias","version-drift"],"status":"partial","priority":"可读","paper_type_zh":"多智能体错误归因数据集、基准与 verifier/reward 方法","best_for_zh":"研究多智能体失败轨迹、SFT/GRPO 归因监督、mixed verifier、数据选择偏差与 episode 回放审计的读者","confidence":"medium","one_line":["Aegis converts deterministic successful multi-agent runs into 9,533 evaluator-confirmed faulty trajectories with agent/error-mode labels, while mixed verifiers and selective failure retention bound reuse.","Aegis 将确定性成功的多智能体运行转化为 9,533 条经 mixed evaluator 确认失败的完整 episode，并附 agent/error-mode 归因；其选择性保留、分组策略未披露与未固定回放限制训练复用。"],"why":"It makes the full MAS episode and injected agent/error map a concrete post-training object for SFT and GRPO, but also shows why synthetic intervention data must expose discarded attempts, grouped splits, verifier versions, baseline lineage, and immutable release manifests.","primary_link":"https://openreview.net/forum?id=zqcYoxXiN3","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kfq20/AEGIS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Fancylalala/AEGIS"},{"key":"project","label":["Project","项目主页"],"url":"https://kfq20.github.io/AEGIS-Website/"}],"link_count":8,"sections":9},{"id":"agent-island-2026","title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","year":2026,"venue":"arXiv preprint","authors":["Connacher Murphy"],"authors_zh":"Connacher Murphy","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["self_play_anchor","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","multiagent_social_strategy","behavioral_analysis"],"tags":["agent-island","multi-agent-benchmark","agent-trajectories","environment-interaction","social-strategy","full-episode-logs","bayesian-plackett-luce","behavioral-audit","live-benchmark","version-drift"],"status":"partial","priority":"可读","paper_type_zh":"多智能体动态 benchmark、环境与 full-episode 数据发布","best_for_zh":"研究 environment/agent trajectory、adaptive-opponent evaluation、行为审计与 live benchmark 版本治理的读者","confidence":"high","one_line":["Agent Island releases a frozen CC BY 4.0 manifest of 999 seven-agent social-strategy episodes and ranks winner outcomes with Bayesian Plackett-Luce, while its separate live leaderboard continues to drift.","Agent Island 发布了含 999 场七智能体社交策略 episode 的冻结 CC BY 4.0 manifest，并以 Bayesian Plackett-Luce 对胜者结果建模；独立的 live leaderboard 会持续漂移。"],"why":"It turns persuasion, alliances, elimination votes, rationales, parser outputs, and final jury decisions into an auditable full-episode benchmark object, but also demonstrates why live counts, matchup effects, generator versions, discarded runs, and training/evaluation boundaries must remain explicit.","primary_link":"https://arxiv.org/abs/2605.04312","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/connachermurphy/agent-island"},{"key":"data","label":["Data","数据"],"url":"https://storage.googleapis.com/agent-island/dataset/agent-island-paper-dataset.json?generation=1778013926769406"},{"key":"project","label":["Project","项目主页"],"url":"https://www.agentisland.ai/"}],"link_count":14,"sections":9},{"id":"agent-diff-benchmarking-llm-agents-on-enterprise-api-tasks-via-code-execution-with-state","title":"Agent-Diff: Benchmarking LLM Agents on Enterprise API Tasks via Code Execution with State-Diff-Based Evaluation","year":2026,"venue":"KDD 2026","authors":["Hubert M. Pysklo","Artem Zhuravel","Patrick D. Watson"],"authors_zh":"Hubert M. Pysklo, Artem Zhuravel, Patrick D. Watson","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Two hundred twenty-four enterprise API tasks scored automatically from database state differences before and after execution.","224 个企业 API 任务用执行前后数据库状态差分自动判分。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2602.11224","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/agent-diff-bench/agent-diff"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hubertmarek/agent-diff-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://agentdiff.dev"}],"link_count":4,"sections":9},{"id":"agent-x-vision-centric-agentic-tasks-2026","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","year":2026,"venue":"ICLR 2026","authors":["Tajamul Ashraf","Amal Saqib","Hanan Gani","Muhra AlMahri","Yuhao Li","Noor Ahsan","Umair Nawaz","Jean Lahoud","Hisham Cholakkal","Mubarak Shah","Philip H. S. Torr","Fahad Shahbaz Khan","Rao Muhammad Anwer","Salman Khan"],"authors_zh":"Tajamul Ashraf、Amal Saqib、Hanan Gani、Muhra AlMahri、Yuhao Li、Noor Ahsan、Umair Nawaz、Jean Lahoud、Hisham Cholakkal、Mubarak Shah、Philip H. S. Torr、Fahad Shahbaz Khan、Rao Muhammad Anwer、Salman Khan","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-agents","tool-use","process-evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测视觉智能体在真实多步任务中如何选择工具、组织过程并完成任务的研究者。","confidence":"high","one_line":["Agent-X provides 828 human-reviewed multimodal tool-use tasks with stepwise traces and judges that score local decisions, full-chain reasoning, and final success separately.","Agent-X 提供 828 个经人工复核的多模态工具使用任务，以逐步轨迹分别评测局部决策、整链推理和最终成功。"],"why":"It exposes agent failures that final task success alone cannot localize, especially where vision context and tool choice interact.","primary_link":"https://openreview.net/forum?id=Vjruxvp1Xd","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mbzuai-oryx/Agent-X"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Tajamul21/Agent-X"}],"link_count":4,"sections":9},{"id":"agentbeats-2026","title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","year":2026,"venue":"arXiv preprint","authors":["Xiaoyuan Liu","Jianhong Tu","Yuqi Chen","Siyuan Xie","Sihan Ren","Tianneng Shi","Gal Gantar","Evan Sandoval","Donghyun Lee","Daniel Miao","Peter J. Gilbert","Nick Hynes","Mauro Staver","Warren He","David Marn","Andrew Low","Xi Zhang","Elron Bandel","Michal Shmueli-Scheuer","Siva Reddy","Alexandre Drouin","Alexandre Lacoste","Ramayya Krishnan","Elham Tabassi","Yu Su","Victor Barres","Chenguang Wang","Wenbo Guo","Dawn Song"],"authors_zh":"Xiaoyuan Liu, Jianhong Tu, Yuqi Chen, Siyuan Xie, Sihan Ren, Tianneng Shi, Gal Gantar, Evan Sandoval, Donghyun Lee, Daniel Miao, Peter J. Gilbert, Nick Hynes, Mauro Staver, Warren He, David Marn, Andrew Low, Xi Zhang, Elron Bandel, Michal Shmueli-Scheuer, Siva Reddy, Alexandre Drouin, Alexandre Lacoste, Ramayya Krishnan, Elham Tabassi, Yu Su, Victor Barres, Chenguang Wang, Wenbo Guo, Dawn Song","tracks":["environment_agent_trajectory_data"],"source_role":["infrastructure","benchmark","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","agent_evaluation","coding_agents"],"tags":["agentbeats","agentified-agent-assessment","agent-evaluation","a2a","mcp","judge-agent","agent-trajectories","environment-feedback","coding-agents","evaluation-infrastructure"],"status":"partial","priority":"可读","paper_type_zh":"智能体评测协议、基础设施与案例研究","best_for_zh":"研究 agent evaluation、环境反馈契约、episode 审计与 replay 边界的读者","confidence":"medium","one_line":["AgentBeats turns a benchmark into an A2A/MCP judge agent that owns tasks, environments and scoring, and validates the interface through a 298-judge/467-subject field study plus standardized coding-agent episodes.","AgentBeats 将 benchmark 封装为掌管任务、环境与评分的 A2A/MCP judge agent，以统一评测 episode 的协议边界；但 arXiv v2 未绑定可重放轨迹、结果与实现清单。"],"why":"It exposes the missing contract around agent trajectories—who provides the environment, what the judge can observe, how terminal outcomes become metrics, and which deployment artifacts must be pinned—while also showing why protocol standardization alone does not make traces licensed or exactly replayable.","primary_link":"https://arxiv.org/abs/2606.13608","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/agentbeats/agentbeats"},{"key":"project","label":["Project","项目主页"],"url":"https://agentbeats.dev/"}],"link_count":9,"sections":9},{"id":"agenther-2026","title":"AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling","year":2026,"venue":"arXiv preprint","authors":["Liang Ding"],"authors_zh":"Liang Ding","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","web_navigation","api_tool_use"],"tags":["agent-trajectories","hindsight-relabeling","failed-rollouts","data-augmentation","llm-as-judge","sft","dpo"],"status":"partial","priority":"可读","paper_type_zh":"智能体失败轨迹后见目标重标与训练数据构造方法","best_for_zh":"研究 agent trajectory 数据构造、LLM judge、SFT/DPO 复用与环境可重放性审计的读者","confidence":"high","one_line":["AgentHER relabels recoverable failed WebArena and ToolBench episodes with LLM-synthesized hindsight goals and packages them as severity-weighted SFT, DPO, or ShareGPT records, while the public code and experiment release remain materially incomplete.","AgentHER 用 LLM 生成的后见目标重标可恢复的 WebArena 与 ToolBench 失败 episode，并封装为 severity-weighted SFT、DPO 或 ShareGPT 记录，但公开代码与实验发布仍存在关键缺口。"],"why":"It treats failed agent interaction logs as reusable post-training data instead of waste and makes the relabeling judge, confidence gate, and failure-severity policy part of the data contract; those same learned components can also introduce false positives, false negatives, and unreproducible selection bias.","primary_link":"https://arxiv.org/abs/2603.21357","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/alphadl/AgentHER"}],"link_count":5,"sections":9},{"id":"agentprocessbench-step-level-tool-agents-2026","title":"AgentProcessBench: Diagnosing Step-Level Process Quality in Tool-Using Agents","year":2026,"venue":"arXiv preprint","authors":["Shengda Fan","Xuyan Ye","Yupeng Huo","Zhi-Yuan Chen","Yiju Guo","Shenzhi Yang","Wenkai Yang","Shuqi Ye","Jingwen Chen","Haotian Chen","Xin Cong","Yankai Lin"],"authors_zh":"Shengda Fan、Xuyan Ye、Yupeng Huo、Zhi-Yuan Chen、Yiju Guo、Shenzhi Yang、Wenkai Yang、Shuqi Ye、Jingwen Chen、Haotian Chen、Xin Cong、Yankai Lin","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["tool-using-agents","process-reward-modeling","agent-evaluation"],"tags":["process-supervision","agent-trajectories","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测工具调用、检索和多轮交互中逐步过程质量及首错定位能力的研究者。","confidence":"high","one_line":["AgentProcessBench supplies 1,000 tool-agent trajectories with 8,509 human ternary step labels for evaluating local process quality and first-error localization.","AgentProcessBench 提供 1,000 条工具智能体轨迹和 8,509 个人工三值步骤标签，用于评测局部过程质量与首错定位。"],"why":"It moves process supervision from closed-world reasoning to tool trajectories where a bad action can have irreversible effects.","primary_link":"https://arxiv.org/abs/2603.14465","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCBM/AgentProcessBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LulaCola/AgentProcessBench"},{"key":"project","label":["Project","项目主页"],"url":"https://rucbm.github.io/AgentProcessBench-Homepage/"}],"link_count":4,"sections":9},{"id":"agentracer-2025","title":"AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems?","year":2026,"venue":"ICLR 2026 Poster","authors":["Guibin Zhang","Junhao Wang","Junjie Chen","Wangchunshu Zhou","Kun Wang","Shuicheng Yan"],"authors_zh":"Guibin Zhang、Junhao Wang、Junjie Chen、Wangchunshu Zhou、Kun Wang、Shuicheng Yan","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","construction_recipe","verifier_reward","process_supervision","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","process_reward"],"training_use":["rlvr","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","multi_agent_systems","environment_interaction","code","mathematics","general_agentic_tasks"],"tags":["environment-agent-trajectory-data","multi-agent-systems","failure-attribution","counterfactual-replay","fault-injection","decisive-step","process-reward","rlvr","release-gap","replay-risk"],"status":"partial","priority":"可读","paper_type_zh":"环境与智能体轨迹数据及失败归因方法","best_for_zh":"研究 agent failure attribution、process supervision、RLVR reward 与轨迹发布审计的读者","confidence":"medium","one_line":["AgenTracer turns successful and failed multi-agent episodes into decisive-agent/step labels through DeepSeek-R1-guided mutation or correction plus environment replay, but only a 127-row coding test subset of the reported 2,476 pairs is public and its license/replay contract is incomplete.","AgenTracer 通过 DeepSeek-R1 引导的纠错或故障注入与环境回放，把多智能体 episode 转成责任智能体/决定性步骤标签；但报告的 2,476 对数据中仅有 127 行 coding test 子集公开，许可与回放契约仍不完整。"],"why":"It separates an environment's terminal pass/fail signal from the causal attribution target used for training a failure tracer, showing both how replay can create grounded process labels and why released trajectories need paired interventions, pinned environments, split lineage, and explicit rights before training reuse.","primary_link":"https://openreview.net/forum?id=l05DseqvuD","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bingreeky/AgenTracer"},{"key":"data","label":["Data","数据"],"url":"https://github.com/bingreeky/AgenTracer/blob/bc6bcd94c2e18f62d6712ca70ba268f6684b894f/data/tracertraj-code-test.parquet"},{"key":"project","label":["Project","项目主页"],"url":"https://bingreeky.github.io/atracer/"}],"link_count":7,"sections":9},{"id":"agentrx-execution-trajectory-diagnosis-2026","title":"AgentRx: Diagnosing AI Agent Failures from Execution Trajectories","year":2026,"venue":"arXiv preprint","authors":["Shraddha Barke","Arnav Goyal","Alind Khare","Avaljot Singh","Suman Nath","Chetan Bansal"],"authors_zh":"Shraddha Barke、Arnav Goyal、Alind Khare、Avaljot Singh、Suman Nath、Chetan Bansal","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent-diagnosis","first-error-localization","constraint-verification"],"tags":["process-supervision","agent-trajectories","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要以关键失败步骤和可追溯证据诊断 API、事件管理与网页／文件智能体轨迹的研究者。","confidence":"high","one_line":["AgentRx provides 115 human-annotated failed executions with a critical step and cross-domain failure category, plus auditable constraint-based diagnosis.","AgentRx 为 115 条失败执行轨迹标注关键失败步骤和跨域失效类别，并提供可审计的约束式诊断记录。"],"why":"Its labels identify the key irreversible step and expose the diagnostic evidence used to justify that decision.","primary_link":"https://arxiv.org/abs/2602.02475","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/AgentRx"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/microsoft/AgentRx"}],"link_count":3,"sections":9},{"id":"agentsim-verifiable-agent-trace-simulation-2026","title":"AgentSim: A Platform for Verifiable Agent-Trace Simulation","year":2026,"venue":"SIGIR 2026","authors":["Saber Zerhoudi","Michael Granitzer","Jelena Mitrović"],"authors_zh":"Saber Zerhoudi, Michael Granitzer, Jelena Mitrović","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["retrieval-augmented-generation","agent-reasoning","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"可验证代理轨迹数据集与 RAG 仿真平台论文","best_for_zh":"需要带检索证据、动作序列与步骤验证信号的 RAG 代理训练或分析研究者","confidence":"high","one_line":["AgentSim releases grounded RAG-agent traces with stepwise decision labels, evidence, and provenance from an active validation pipeline.","发布 10 万余条带证据与决策标签的检索增强代理步骤，可训练 grounded 轨迹模型和过程奖励模型。"],"why":"The release preserves the evidence and validation context required to study grounded agent process supervision rather than only terminal answers.","primary_link":"https://arxiv.org/abs/2604.26653","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/searchsim-org/sigir26-agentsim"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/searchsim/agentsim-atc"}],"link_count":5,"sections":9},{"id":"agentsynth-generalist-computer-use-2026","title":"AgentSynth: Scalable Task Generation for Generalist Computer-Use Agents","year":2026,"venue":"ICLR 2026","authors":["Jingxu Xie","Dylan Xu","Xuandong Zhao","Dawn Song"],"authors_zh":"Jingxu Xie, Dylan Xu, Xuandong Zhao, Dawn Song","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["computer-use","gui-agents","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"通用计算机使用智能体的任务合成与长程轨迹数据集论文","best_for_zh":"需要可控任务难度、真实桌面操作轨迹和低成本合成机制的计算机使用智能体研究者。","confidence":"high","one_line":["AgentSynth releases verified computer-use task sequences and trajectories whose difficulty scales by composing simple subtasks.","AgentSynth 将可解的简单子任务链式组合为难度可控的长程计算机使用任务，并公开相应执行轨迹。"],"why":"It provides a practical, openly reproducible mechanism for constructing long-horizon computer-use supervision at controlled difficulty.","primary_link":"https://openreview.net/pdf?id=CoBxmXThM6","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sunblaze-ucb/AgentSynth"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sunblaze-ucb/AgentSynth"},{"key":"project","label":["Project","项目主页"],"url":"https://sunblaze-ucb.github.io/agentsynth_web/"}],"link_count":7,"sections":9},{"id":"plan-rewardbench-trajectory-reward-modeling-2026","title":"Aligning Agents via Planning: A Benchmark for Trajectory-Level Reward Modeling","year":2026,"venue":"ACL 2026","authors":["Jiaxuan Wang","Yulan Hu","Wenjin Yang","Zheng Pan","Xin Li","Lan-Zhe Guo"],"authors_zh":"Jiaxuan Wang, Yulan Hu, Wenjin Yang, Zheng Pan, Xin Li, Lan-Zhe Guo","tracks":["preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agents","trajectory-preference","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"工具智能体长轨迹偏好比较与奖励模型评测基准论文","best_for_zh":"需要在规划、工具调用、安全拒答和错误恢复中训练或诊断轨迹级奖励模型的研究者。","confidence":"high","one_line":["Plan-RewardBench evaluates whether a judge can prefer safe, recoverable tool-use trajectories over closely confusable alternatives.","1,171 个轨迹偏好比较、七个切分，覆盖规划、工具失效恢复、安全拒答与无关工具识别，适合长程奖励研究。"],"why":"It shifts feedback from isolated answers to long-horizon agent decisions.","primary_link":"https://aclanthology.org/2026.acl-long.1062/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wyy-1112/Plan-RewardBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/wyy1112/Plan-RewardBench"}],"link_count":5,"sections":9},{"id":"alignment-data-map-selection-2026","title":"Alignment Data Map for Efficient Preference Data Selection and Diagnosis","year":2026,"venue":"Findings of ACL 2026","authors":["Seohyeong Lee","Eunwon Kim","Hwaran Lee","Buru Chang"],"authors_zh":"Seohyeong Lee, Eunwon Kim, Hwaran Lee, Buru Chang（西江大学、Upstage、高丽大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-data","data-selection","label-diagnosis","dpo","simpo"],"status":"verified","priority":"必读","paper_type_zh":"偏好数据选择与标注诊断研究","best_for_zh":"适合在 DPO 或 SimPO 前选择或审计偏好数据、且不满足于只看奖励间隔的读者。","confidence":"high","one_line":["Alignment Data Map selects preference records by jointly measuring response quality and within-prompt variability, while exposing likely label mismatches.","Alignment Data Map 联合衡量回答质量与同题候选波动，选择更有效的偏好记录并暴露可能错标的样本。"],"why":"It makes absolute response quality and ambiguity jointly visible before a preference record is sent to training.","primary_link":"https://aclanthology.org/2026.findings-acl.1906/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/01choco/Alignment-Data-Map"}],"link_count":3,"sections":9},{"id":"amo-bench-olympiad-math-2026","title":"AMO-Bench: Large Language Models Still Struggle in High School Math Competitions","year":2026,"venue":"ACL 2026","authors":["Shengnan An","Xunliang Cai","Xuezhi Cao","Xiaoyu Li","Yehao Lin","Junlin Liu","Xinxuan Lv","Dan Ma","Xuanlin Wang","Ziwen Wang","Shuang Zhou"],"authors_zh":"Shengnan An、Xunliang Cai、Xuezhi Cao、Xiaoyu Li、Yehao Lin、Junlin Liu、Xinxuan Lv、Dan Ma、Xuanlin Wang、Ziwen Wang、Shuang Zhou","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"必读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["AMO-Bench uses high-school olympiad problems to expose persistent gaps in rigorous mathematical reasoning.","用高中奥数竞赛题揭示前沿模型在严谨数学推理上的持续短板。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://arxiv.org/abs/2510.26768","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/meituan-longcat/AMO-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/meituan-longcat/AMO-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://amo-bench.github.io/"}],"link_count":4,"sections":9},{"id":"traceeval-2026","title":"An Execution-Verified Multi-Language Benchmark for Code Semantic Reasoning","year":2026,"venue":"arXiv preprint","authors":["Yikun Li","Jinfeng Jiang","Ting Zhang","Chengran Yang","Chenxing Zhong","Yin Yide","Leow Wen Bin","Eng Lieh Ouh","Lwin Khin Shar","David Lo"],"authors_zh":"Yikun Li 等（Singapore Management University、Monash University、Nanjing University of Science and Technology、GovTech Singapore）","tracks":["benchmarks_evaluation_surfaces","programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit","sft"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["code-semantics","code-executable-benchmark"],"tags":["benchmark","code-semantics","code_executable_benchmark","evaluation-surface"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的评测面 / 环境基准","best_for_zh":"关注代码语义评测、运行验证标签、调用图恢复和代码数据复用边界的读者。","confidence":"medium","one_line":["TraceEval evaluates code semantic reasoning by asking models to recover runtime call structure whose positive edges are witnessed by validation execution.","TraceEval 通过运行验证过的调用边，评测模型能否从源码恢复程序运行时调用结构。"],"why":"TraceEval evaluates code semantic reasoning by asking models to recover runtime call structure whose positive edges are witnessed by validation execution.","primary_link":"https://arxiv.org/abs/2605.11006","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yikun-li/TraceEva"}],"link_count":3,"sections":9},{"id":"an-imperfect-verifier-is-good-enough-learning-with-noisy-rewards-2026","title":"An Imperfect Verifier is Good Enough: Learning with Noisy Rewards","year":2026,"venue":"arXiv preprint arXiv:2604.07666","authors":["Andreas Plesner","Francisco Guzmán","Anish Athalye"],"authors_zh":"Andreas Plesner、Francisco Guzmán、Anish Athalye","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🪜 Process Supervision and Process Reward Models map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","研究含噪验证器是否仍可支持有效学习，直接关系到 RLVR 中误判奖励会怎样影响策略更新。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://arxiv.org/abs/2604.07666","links":[],"link_count":1,"sections":9},{"id":"anchor-branch-point-data-generation-gui-agents-2026","title":"ANCHOR: Branch-Point Data Generation for GUI Agents","year":2026,"venue":"ACL 2026","authors":["Jinbiao Wei","Yilun Zhao","Kangqi Ni","Arman Cohan"],"authors_zh":"Jinbiao Wei, Yilun Zhao, Kangqi Ni, Arman Cohan","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["gui-agents","computer-use","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"GUI 智能体分叉轨迹扩展与步骤级过滤数据集论文","best_for_zh":"需要在桌面 GUI 环境中构造、筛选或训练具备过程纠错能力的智能体轨迹的研究者。","confidence":"high","one_line":["ANCHOR releases branch-point-expanded GUI trajectories with state-aware verification and step-level denoising.","ANCHOR 从已验证 GUI 演示中的分叉点扩展任务与轨迹，再用状态感知校验和步骤过滤生成面向局部决策的监督数据。"],"why":"It targets the local decision point at which GUI trajectories diverge, rather than labeling only completed episodes.","primary_link":"https://arxiv.org/abs/2602.07153","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yale-nlp/Anchor"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yale-nlp/Anchor"}],"link_count":5,"sections":9},{"id":"arise-tts-evaluation-2026","title":"ARISE: An Adaptive Resolution-Aware Metric for Test-Time Scaling Evaluation in Large Reasoning Models","year":2026,"venue":"Findings of ACL 2026","authors":["Zhangyue Yin","Qiushi Sun","Zhiyuan Zeng","Zhiyuan Yu","Qipeng Guo","Xuanjing Huang","Xipeng Qiu"],"authors_zh":"Zhangyue Yin、Qiushi Sun、Zhiyuan Zeng、Zhiyuan Yu、Qipeng Guo、Xuanjing Huang、Xipeng Qiu（机构：复旦大学、香港大学、南京大学、上海 AI Laboratory、上海创新研究院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","code-generation","agentic-tasks"],"tags":["test-time-compute","evaluation","scaling-efficiency","token-cost","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"测试时扩展评测指标研究（Findings of ACL 2026）","best_for_zh":"比较推理模型质量—成本扩展行为的读者。","confidence":"high","one_line":["ARISE evaluates test-time scaling by rewarding per-sample improvements and penalizing accuracy regressions that consume more tokens.","ARISE 以样本级正确性变化和词元成本评测测试时扩展，并惩罚越算越错的负向扩展。"],"why":"It prevents a model from appearing efficient when it spends extra tokens to turn correct answers into errors.","primary_link":"https://aclanthology.org/2026.findings-acl.289/","links":[],"link_count":2,"sections":9},{"id":"aa-intelligence-index-2026","title":"Artificial Analysis Intelligence Index v4.1","year":2026,"venue":"Artificial Analysis methodology","authors":["Artificial Analysis"],"authors_zh":"Artificial Analysis","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["mixed","judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit","reward_verifier_layer"],"domains":["model-evaluation","aggregate-intelligence-index","agentic-and-coding-evaluation"],"tags":["benchmark","aggregate-evaluation","model-evaluation"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark methodology / aggregate index","best_for_zh":"需要审计聚合 benchmark、组件权重、混合 verifier 和 leaderboard 可复现边界的读者。","confidence":"medium","one_line":["Artificial Analysis Intelligence Index aggregates multiple evaluations into a weighted model-quality score.","Artificial Analysis 智能指数把多个评测按权重合成为单一的模型质量分数。"],"why":"It is useful as a benchmark-of-benchmarks example where the aggregate number is less reusable than its component contracts.","primary_link":"https://artificialanalysis.ai/methodology/intelligence-benchmarking","links":[{"key":"project","label":["Project","项目主页"],"url":"https://artificialanalysis.ai/"}],"link_count":2,"sections":9},{"id":"audio-multichallenge-spoken-dialogue-2026","title":"Audio MultiChallenge: A Multi-Turn Evaluation of Spoken Dialogue Systems on Natural Human Interaction","year":2026,"venue":"ACL 2026","authors":["Advait Gosai","Tyler Vuong","Utkarsh Tyagi","Steven Li","Wenjia You","Miheer Bavare","Arda Uçar","Zhongwang Fang","Brian Jang","Bing Liu","Yunzhong He"],"authors_zh":"Advait Gosai、Tyler Vuong、Utkarsh Tyagi、Steven Li、Wenjia You、Miheer Bavare、Arda Uçar、Zhongwang Fang、Brian Jang、Bing Liu、Yunzhong He","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["Audio MultiChallenge evaluates spoken dialogue systems on natural multi-turn interaction, including audio-native memory and voice editing.","以自然人声中的记忆、约束保持、自洽与语音编辑检验端到端语音对话系统。"],"why":"Audio MultiChallenge evaluates spoken dialogue systems on natural multi-turn interaction, including audio-native memory and voice editing.","primary_link":"https://aclanthology.org/2026.acl-long.1654/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScaleAI/audiomc"},{"key":"project","label":["Project","项目主页"],"url":"https://labs.scale.com/leaderboard/audiomc"}],"link_count":3,"sections":9},{"id":"auto-l2s-2025","title":"AutoL2S: Auto Long-Short Reasoning for Efficient Large Language Models","year":2026,"venue":"Findings of ACL 2026","authors":["Feng Luo","Yu-Neng Chuang","Guanchu Wang","Hoang Anh Duy Le","Shaochen Zhong","Hongyi Liu","Jiayi Yuan","Yang Sui","Vladimir Braverman","Vipin Chaudhary","Xia Hu"],"authors_zh":"Feng Luo；Yu-Neng Chuang；Guanchu Wang；Hoang Anh Duy Le；Shaochen Zhong；Hongyi Liu；Jiayi Yuan；Yang Sui；Vladimir Braverman；Vipin Chaudhary；Xia Hu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["distillation","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["long-to-short","distillation","reasoning-compression","grpo","shortest-correct"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A long-to-short reasoning-distillation recipe that attempts to retain correct concise traces.","AutoL2S 将长推理与短推理配对，保留正确且更短的答案，并以长度控制标记训练可在长短推理间切换的模型。"],"why":"It is a Track 5 bridge from expensive long rollouts to compact trainable traces, but the unreleased pair corpus limits independent audit.","primary_link":"https://aclanthology.org/2026.findings-acl.831/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/amandaluof/AutoL2S"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/amandaa/AutoL2S-7b"}],"link_count":4,"sections":9},{"id":"autorubric-rubric-based-evaluation-2026","title":"Autorubric: Unifying Rubric-based LLM Evaluation","year":2026,"venue":"arXiv 2026","authors":["Delip Rao","Chris Callison-Burch"],"authors_zh":"Delip Rao、Chris Callison-Burch","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","sft","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["llm_judge","rubric_evaluation"],"tags":["rubric","llm_judge","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测框架与数据论文","best_for_zh":"需要构建可校准 rubric Judge 或复用 CHARM-100 类评测数据的研究者。","confidence":"high","one_line":["Autorubric unifies calibrated, rubric-based LLM evaluation with mixed criterion types, ensembles, and released benchmark artifacts.","统一二元、序数与名义 rubric 的校准、集成评审和可靠性度量，并发布 CHARM-100 等数据制品。"],"why":"It exposes rubric-grounded evaluation signals and their audit boundary.","primary_link":"https://arxiv.org/abs/2603.00077","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/delip/autorubric"},{"key":"project","label":["Project","项目主页"],"url":"https://autorubric.org/"}],"link_count":4,"sections":9},{"id":"autosuit-bench-security-unit-tests-2026","title":"AutoSUIT Bench: Automated Security UnIt Test Benchmark for LLM Coding","year":2026,"venue":"Findings of ACL 2026","authors":["Samuel Osebe","Fan Yang","Junyi Li","Yue Gu","Yongxin Wang","Satyapriya Krishna","Kai-Wei Chang","Aram Galstyan","Rahul Gupta","Weitong Ruan"],"authors_zh":"Samuel Osebe, Fan Yang, Junyi Li, Yue Gu, Yongxin Wang, Satyapriya Krishna, Kai-Wei Chang, Aram Galstyan, Rahul Gupta, Weitong Ruan","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-security","unit-tests"],"tags":["programmatic-verification","benchmark","2026"],"status":"verified","priority":"必读","paper_type_zh":"安全单元测试代码生成基准","best_for_zh":"需要评测或训练兼顾功能正确性与安全测试覆盖的代码模型的研究者。","confidence":"high","one_line":["AutoSUIT Bench releases security-oriented unit-test suites spanning 232 CWE classes to jointly score functional and security behavior of generated code.","AutoSUIT Bench 自动构造覆盖 232 类 CWE 的安全单元测试套件，以编译和测试断言共同衡量代码功能与安全行为。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://aclanthology.org/2026.findings-acl.1735/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/amazon/AutoSUIT"}],"link_count":2,"sections":9},{"id":"barrierbench-evaluating-large-language-models-for-safety-verification-in-dynamical-syste","title":"BarrierBench: Evaluating Large Language Models for Safety Verification in Dynamical Systems","year":2026,"venue":"L4DC 2026","authors":["Ali Taheri","Alireza Taban","Sadegh Soudjani","Ashutosh Trivedi"],"authors_zh":"Ali Taheri, Alireza Taban, Sadegh Soudjani, Ashutosh Trivedi","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** One hundred dynamical-system barrier-certificate tasks strictly verify LLM-generated safety certificates with SMT.","100 个动力系统 barrier certificate 任务，用 SMT 严格验证 LLM 生成安全证书。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://proceedings.mlr.press/v331/","links":[{"key":"data","label":["Data","数据"],"url":"https://hycodev.com/dataset/barrierbench"}],"link_count":3,"sections":9},{"id":"general-agentbench-test-time-scaling-2026","title":"Benchmark Test-Time Scaling of General LLM Agents","year":2026,"venue":"arXiv preprint","authors":["Xiaochuan Li","Ryan Ming","Pranav Setlur","Abhijay Paladugu","Andy Tang","Hao Kang","Shuai Shao","Rong Jin","Chenyan Xiong"],"authors_zh":"Xiaochuan Li、Ryan Ming、Pranav Setlur、Abhijay Paladugu、Andy Tang、Hao Kang、Shuai Shao、Rong Jin、Chenyan Xiong（机构：卡内基梅隆大学、Meta）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-reasoning","information-seeking","mathematical-reasoning","software-engineering"],"tags":["test-time-scaling","agents","parallel-sampling","sequential-scaling","verification","benchmark"],"status":"verified","priority":"可读","paper_type_zh":"测试时扩展基准与诊断研究","best_for_zh":"需要决定是否为工具智能体增加交互轮次或并行轨迹，并设计相应验证器的读者。","confidence":"high","one_line":["General AgentBench evaluates sequential and parallel test-time scaling across general LLM agents and finds that extra trajectories help only when context use and trajectory verification remain reliable.","General AgentBench 在通用大语言模型智能体上评测顺序与并行两类测试时扩展，并发现额外轨迹只有在上下文利用和轨迹验证可靠时才会带来收益。"],"why":"It shows that a larger inference budget alone does not reliably scale general agents: sequential runs hit a context ceiling, while parallel runs expose a verification gap.","primary_link":"https://arxiv.org/abs/2602.18998","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/cxcscmu/General-AgentBench"}],"link_count":2,"sections":9},{"id":"prmbench-v-2026","title":"Benchmarking Fine-Grained Error Detection in Multimodal Reasoning","year":2026,"venue":"ACL 2026","authors":["Chi-Min Chan","Han Zhu","Chunyang Jiang","Jiaming Ji","Juntao Dai","Wei Xue","Sirui Han","Yike Guo"],"authors_zh":"Chi-Min Chan, Han Zhu, Chunyang Jiang, Jiaming Ji, Juntao Dai, Wei Xue, Sirui Han, Yike Guo","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"多模态过程奖励模型评测与错误诊断论文","best_for_zh":"需要选择或审计视觉推理过程裁判，并评估其 Best-of-N 选择价值的研究者。","confidence":"medium","one_line":["PRMBench-V provides human-verified step-level error categories for auditing multimodal process reward models.","以人工核验的九类局部错误，评估多模态过程奖励模型是否真正识别视觉推理失误。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.acl-long.2068/","links":[],"link_count":1,"sections":9},{"id":"longjudgebench-2026","title":"Benchmarking LLM-as-a-Judge for Long-Form Output Evaluation","year":2026,"venue":"arXiv preprint","authors":["Junjie Chen","Yuxi Dong","Haitao Li","Weihang Su","Yujia Zhou","Min Zhang","Yiqun Liu","Qinyao Ai"],"authors_zh":"Junjie Chen, Yuxi Dong, Haitao Li, Weihang Su, Yujia Zhou, Min Zhang, Yiqun Liu, Qinyao Ai","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要评测报告、计划或长回答的自动评审器研究者。","confidence":"medium","one_line":["Long-form judge benchmark emphasizing protocol-dependent reliability.","以跨场景、跨协议基准检验 LLM 评审器对长文本输出的稳定性。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://arxiv.org/abs/2606.01629","links":[{"key":"data","label":["Data","数据"],"url":"https://arxiv.org/pdf/2606.01629"}],"link_count":2,"sections":9},{"id":"unleakedtestbench-real-world-unit-test-generation-2026","title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","year":2026,"venue":"ACM Transactions on Software Engineering and Methodology","authors":["Dong Huang","Jie M. Zhang","Mark Harman","Qianru Zhang","Mingzhe Du","See-Kiong Ng"],"authors_zh":"Dong Huang, Jie M. Zhang, Mark Harman, Qianru Zhang, Mingzhe Du, See-Kiong Ng","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit","sft","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-testing"],"tags":["unit-test-generation","benchmark-contamination","python","mutation-testing","executable-evaluation","tosem-2026"],"status":"verified","priority":"可读","paper_type_zh":"泄漏控制的单元测试生成基准","best_for_zh":"研究测试生成、代码泛化或基准污染的研究者。","confidence":"high","one_line":["UnLeakedTestBench pairs 3,909 leakage-controlled real Python functions with a leaked counterpart and execution, coverage, and mutation metrics to separate test-generation generalization from memorization.","UnLeakedTestBench 以 3,909 个泄漏受控的真实 Python 函数及配对泄漏集，用执行、覆盖率和变异测试区分测试生成的泛化与记忆。"],"why":"It exposes when strong test-generation scores come from leaked benchmark knowledge rather than robust testing of real, complex functions.","primary_link":"https://arxiv.org/abs/2508.00408","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/huangd1999/UnLeakedTestBench"}],"link_count":4,"sections":9},{"id":"best-of-majority-passk-2026","title":"Best-of-Majority: Minimax-Optimal Strategy for Pass@k Inference Scaling","year":2026,"venue":"ICLR 2026","authors":["Qiwei Di","Kaixuan Ji","Xuheng Li","Heyang Zhao","Quanquan Gu"],"authors_zh":"Qiwei Di、Kaixuan Ji、Xuheng Li、Heyang Zhao、Quanquan Gu（加州大学洛杉矶分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","pass-at-k","inference-scaling","reward-model"],"status":"verified","priority":"必读","paper_type_zh":"测试时推理扩展理论与数学推理评测论文（ICLR 2026）","best_for_zh":"设计 pass@k 评测、奖励模型选择或有限推理预算分配的研究者。","confidence":"high","one_line":["Best-of-Majority gives a minimax-optimal, scaling-monotonic way to allocate sampled candidates under a Pass@k inference budget.","Best-of-Majority 在有限采样和提交预算下，以频率筛选结合奖励排序实现可随样本数稳定扩展的 pass@k 推理策略。"],"why":"It separates the sampling budget, submission budget, and reward-model error rather than treating more samples as an automatic gain.","primary_link":"https://arxiv.org/abs/2510.03199","links":[],"link_count":3,"sections":9},{"id":"crystal-transparent-multimodal-reasoning-2026","title":"Beyond Final Answers: CRYSTAL Benchmark for Transparent Multimodal Reasoning Evaluation","year":2026,"venue":"ECCV 2026","authors":["Wayner Barrios","SouYoung Jin"],"authors_zh":"Wayner Barrios、SouYoung Jin","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","process-reward-modeling","benchmarking"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要以可验证中间步骤评测多模态推理，并把答案正确性与推理覆盖度、顺序分开分析的研究者。","confidence":"high","one_line":["CRYSTAL supplies 6,372 quality-gated multimodal reference chains and order-sensitive metrics that expose correct answers reached through incomplete or disordered reasoning.","CRYSTAL 提供 6,372 条经质量门控的多模态参考推理链，并以顺序敏感指标揭示“答对但推理缺失或失序”的情况。"],"why":"It makes step coverage and ordering observable in multimodal evaluation instead of treating a correct final answer as sufficient evidence of reasoning.","primary_link":"https://arxiv.org/abs/2603.13099","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/waybarrios/crystal"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/waybarrios/CRYSTAL"}],"link_count":3,"sections":9},{"id":"trajfusion-2026","title":"Beyond Rejection Sampling: Trajectory Fusion for Scaling Mathematical Reasoning","year":2026,"venue":"Findings of ACL 2026","authors":["Jie Deng","Hanshuang Tong","Jun Li","Shining Liang","Ning Wu","Hongzhi Li","Yutao Xie"],"authors_zh":"Jie Deng、Hanshuang Tong、Jun Li、Shining Liang、Ning Wu、Hongzhi Li、Yutao Xie","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer"],"domains":["mathematics"],"tags":["trajectory-fusion","rejection-sampling","incorrect-trajectories","reflection-prompts","error-diversity","mathematical-reasoning","answer-verification","synthetic-data","sft"],"status":"partial","priority":"可读","paper_type_zh":"错误轨迹融合构造配方","best_for_zh":"研究拒绝采样、错误轨迹复用、数学 SFT 与过程审计的读者","confidence":"high","one_line":["TrajFusion samples 16 math solutions, selects representative verifier-rejected errors using frequency and wrong-answer diversity, and fuses them with generic reflection prompts and one correct trace for standard SFT.","TrajFusion 从 16 条数学解答中选择代表性错误，将其与通用反思和一条正确轨迹融合为标准 SFT 目标；价值在构造配方，而非尚未发布的数据复用。"],"why":"It gives a concrete alternative to discarding every wrong rollout, while making clear that terminal correctness, generic reflection text, and an unreleased candidate pool leave substantial process-quality and lineage risks.","primary_link":"https://aclanthology.org/2026.findings-acl.390/","links":[],"link_count":4,"sections":9},{"id":"beyond-theorem-proving-formal-problem-solving-2026","title":"Beyond Theorem Proving: Formulation, Framework and Benchmark for Formal Problem-Solving","year":2026,"venue":"ICML 2026 Spotlight","authors":["Qi Liu","Xinhao Zheng","Renqiu Xia","Xingzhi Qi","Qinxiang Cao","Junchi Yan"],"authors_zh":"Qi Liu, Xinhao Zheng, Renqiu Xia, Xingzhi Qi, Qinxiang Cao, Junchi Yan","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["evaluation","rlvr","agent_training"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4"],"tags":["lean4","formal-problem-solving","rlvr","benchmark","icml-2026"],"status":"verified","priority":"可读","paper_type_zh":"可执行验证的形式问题求解基准","best_for_zh":"研究 Lean 推理、形式强化学习或答案等价验证的研究者。","confidence":"high","one_line":["FPS turns unknown-answer mathematics into replayable Lean state transitions, with RPE checking whether a completed formal answer is equivalent to the target.","FPS 将未知答案的数学求解建模为 Lean 中可重放的状态转移，并用 RPE 验证完成答案与目标的等价性。"],"why":"It evaluates answer discovery and proof jointly instead of treating the answer as already given to the prover.","primary_link":"https://arxiv.org/abs/2505.04528","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Purewhite2019/formal_problem_solving_main"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/purewhite42/formal-problem-solving"}],"link_count":4,"sections":9},{"id":"cobalt-reward-hacking-2026","title":"Bridging Online and Offline RL: Contextual Bandit Learning for Multi-Turn Code Generation","year":2026,"venue":"Second Conference on Language Modeling, 2025","authors":["Ziru Chen","Dongdong Chen","Ruinan Jin","Yingbin Liang","Yujia Xie","Huan Sun"],"authors_zh":"Ziru Chen, Dongdong Chen, Ruinan Jin, Yingbin Liang, Yujia Xie, Huan Sun","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"多轮代码强化学习与奖励投机审计论文","best_for_zh":"需要降低多轮代码 RL 成本、分析错误执行反馈或训练抗奖励投机代理的研究者。","confidence":"high","one_line":["perturbed trajectory data for in-context reward-hacking analysis","将多轮代码 RL 化为离线轨迹上的上下文 bandit 学习，并用扰动轨迹缓解错误反馈诱发的奖励投机。"],"why":"It provides a concrete audit surface for erroneous-feedback reward hacking in multi-turn code RL.","primary_link":"https://openreview.net/forum?id=iB4H079VV0","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OSU-NLP-Group/cobalt"}],"link_count":3,"sections":9},{"id":"browsecomp-plus-2025","title":"BrowseComp-Plus: A Fair and Disentangled Evaluation Benchmark for Deep Search Agents","year":2026,"venue":"ACL 2026","authors":["Zijian Chen","Xueguang Ma","Shengyao Zhuang","Ping Nie","Kai Zou","Sahel Sharifymoghaddam","Andrew Liu","Joshua Green","Kshama Patel","Ruoxi Meng","Mingyi Su","Yanxi Li","Haoran Hong","Xinyu Shi","Xuye Liu","Hosna Oyarhoseini","Nandan Thakur","Crystina Zhang","Luyu Gao","Wenhu Chen","Jimmy Lin"],"authors_zh":"Zijian Chen, Xueguang Ma, Shengyao Zhuang, Ping Nie, Kai Zou, Sahel Sharifymoghaddam, Andrew Liu, Joshua Green, Kshama Patel, Ruoxi Meng, Mingyi Su, Yanxi Li, Haoran Hong, Xinyu Shi, Xuye Liu, Hosna Oyarhoseini, Nandan Thakur, Crystina Zhang, Luyu Gao, Wenhu Chen, Jimmy Lin","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release","agent_environment","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","web_search","information_retrieval","question_answering","deep_research"],"tags":["browsecomp-plus","deep-search","web-agent","retrieval-agent","fixed-corpus","agent-trajectory","qrels","citation-evaluation","llm-as-judge","benchmark-contamination"],"status":"partial","priority":"可读","paper_type_zh":"固定语料深度检索智能体评测基准与部分轨迹发布","best_for_zh":"研究 web/search agent、检索反馈、轨迹重放、judge 漂移与 benchmark 审计的读者","confidence":"medium-high","one_line":["A fixed-corpus BrowseComp derivative with 830 answerable queries, 100,195 web documents, human evidence/gold qrels, mixed answer/retrieval/citation grading, and four released frontier-agent trajectory archives.","BrowseComp-Plus 将 BrowseComp 筛为 830 个可核验证据的查询，并配套 100,195 篇固定网页语料、人类 evidence/gold qrels、混合评测器与四组部分发布的完整 agent 轨迹。"],"why":"It disentangles the search agent from live-web drift and exposes whether failures arise from retrieval, citation, terminal status, or answer synthesis. The fixed substrate and partial episode release support replay and audit, while judge drift, incomplete qrels, public reversible obfuscation, and unresolved web-content licensing remain important limits.","primary_link":"https://aclanthology.org/2026.acl-long.1023/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/texttron/BrowseComp-Plus"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Tevatron/browsecomp-plus"},{"key":"project","label":["Project","项目主页"],"url":"https://texttron.github.io/BrowseComp-Plus/"}],"link_count":12,"sections":9},{"id":"browsecomp-v3-multimodal-browsing-2026","title":"BrowseComp-V3: A Visual, Vertical, and Verifiable Benchmark for Multimodal Browsing Agents","year":2026,"venue":"arXiv preprint","authors":["Huanyao Zhang","Jiepeng Zhou","Bo Li","Bowen Zhou","Yanzhe Shan","Haishan Lu","Zhiyong Cao","Jiaoyang Chen","Yuqian Han","Zinan Sheng","Zhengwei Tao","Hao Liang","Jialong Wu","Yang Shi","Yuanpeng He","Jiaye Lin","Qintong Zhang","Guochen Yan","Runhao Zhao","Zhengpin Li","Xiaohan Yu","Lang Mei","Chong Chen","Wentao Zhang","Bin Cui"],"authors_zh":"Huanyao Zhang 等（Peking University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","multimodal-browsing","vertical-search"],"tags":["agent_environment","trajectory_data","web-browser-agents","multimodal-browsing","vertical-search"],"status":"verified","priority":"可读","paper_type_zh":"arXiv 的 multimodal browsing benchmark","best_for_zh":"关注 Web、GUI、移动端、OS 智能体环境、轨迹数据和可复现评测的研究者。","confidence":"medium","one_line":["BrowseComp-V3 specifies visual, vertical, verifiable tasks for multimodal browsing agents.","BrowseComp-V3 定义面向多模态浏览智能体的视觉、垂直、可核验任务。"],"why":"it extends browsing-agent evaluation toward visual and vertical information seeking","primary_link":"https://arxiv.org/abs/2602.12876","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Halcyon-Zhang/BrowseComp-V3"}],"link_count":3,"sections":9},{"id":"browseconf-2026","title":"BrowseConf: Confidence-Guided Test-Time Scaling for Web Agents","year":2026,"venue":"Findings of ACL 2026","authors":["Litu Ou","Kuan Li","Huifeng Yin","Liwen Zhang","Zhongwang Zhang","Xixi Wu","Rui Ye","Zile Qiao","Yong Jiang","Pengjun Xie","Fei Huang","Jingren Zhou"],"authors_zh":"Litu Ou、Kuan Li、Huifeng Yin、Liwen Zhang、Zhongwang Zhang、Xixi Wu、Rui Ye、Zile Qiao、Yong Jiang、Pengjun Xie、Fei Huang、Jingren Zhou","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["web_agents","information_seeking","reasoning"],"tags":["web-agent","confidence","rollout","early-stopping"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["BrowseConf uses verbalized confidence to stop/retry web-agent rollouts.","BrowseConf 用模型自述的置信度决定网页智能体的采样何时停止、何时重试。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"A per-rollout trace field controls interactive allocation and memory.","primary_link":"https://aclanthology.org/2026.findings-acl.21/","links":[],"link_count":4,"sections":9},{"id":"can-ai-agents-answer-your-data-questions-a-benchmark-for-data-agents","title":"Can AI Agents Answer Your Data Questions? A Benchmark for Data Agents","year":2026,"venue":"arXiv","authors":["Ruiying Ma","Shreya Shankar","Ruiqi Chen","Yiming Lin","Sepanta Zeighami","Rajoshi Ghosh","Abhinav Gupta","Anushrut Gupta","Tanmai Gopal","Aditya G. Parameswaran"],"authors_zh":"Ruiying Ma, Shreya Shankar, Ruiqi Chen, Yiming Lin, Sepanta Zeighami, Rajoshi Ghosh, Abhinav Gupta, Anushrut Gupta, Tanmai Gopal, Aditya G. Parameswaran","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Fifty-four realistic multi-database agent questions spanning 12 datasets, nine domains, and four DBMSes.","54 个跨 12 数据集、9 领域、4 DBMS 的真实多数据库 agent 问题。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2603.20576","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ucbepic/DataAgentBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ruiyingm/DataAgentBench-data"},{"key":"project","label":["Project","项目主页"],"url":"https://ucbepic.github.io/DataAgentBench/"}],"link_count":4,"sections":9},{"id":"ruverbench-rubric-verification-agents-2026","title":"Can LLM-as-a-Judge Reliably Verify Rubrics in Agentic Scenarios?","year":2026,"venue":"arXiv 2026","authors":["Yangda Peng","Yunjia Qi","Hao Peng","Haotian Xia","Guanzhong He","Xintong Shi","Richeng Xuan","Songyuanyi Lu","Yixian Liu","Zhichao Hu","Yuhong Liu","Lei Hou","Bin Xu","Juanzi Li"],"authors_zh":"Yangda Peng、Yunjia Qi、Hao Peng、Haotian Xia、Guanzhong He、Xintong Shi、Richeng Xuan、Songyuanyi Lu、Yixian Liu、Zhichao Hu、Yuhong Liu、Lei Hou、Bin Xu、Juanzi Li","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["llm_judge","rubric_evaluation"],"tags":["rubric","llm_judge","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"Judge 元评测基准论文","best_for_zh":"审计深度研究或代码 Agent 的 rubric 评审可靠性的研究者。","confidence":"high","one_line":["RuVerBench tests whether LLM judges can verify rubric compliance for long agentic reports and code against human labels.","以 2,458 条长报告或代码输出、rubric 和人工满足性标签检验 Agent 场景 Judge 的可靠性。"],"why":"It exposes rubric-grounded evaluation signals and their audit boundary.","primary_link":"https://arxiv.org/abs/2606.29920","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/RuVerBench"}],"link_count":3,"sections":9},{"id":"cats-calibrated-ttc-2026","title":"CaTS: Calibrated Test-Time Scaling for Efficient LLM Reasoning","year":2026,"venue":"ICLR 2026","authors":["Chengsong Huang","Langlin Huang","Jixuan Leng","Jiacheng Liu","Jiaxin Huang"],"authors_zh":"Chengsong Huang、Langlin Huang、Jixuan Leng、Jiacheng Liu、Jiaxin Huang（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["test_time_compute","distillation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning","mathematics"],"tags":["test-time-compute","calibration","adaptive-sampling","self-consistency"],"status":"verified","priority":"必读","paper_type_zh":"校准式自适应测试时扩展论文（ICLR 2026）","best_for_zh":"设计置信度引导采样或推理预算评测的研究者。","confidence":"high","one_line":["CaTS distills self-consistency confidence to allocate repeated-sampling budget and stop easy queries early.","CaTS 将自一致性得到的置信度蒸馏为可校准信号，用于在固定平均预算下动态停止或分配重复采样。"],"why":"It makes the confidence estimator and the sample budget part of the same reproducible inference contract.","primary_link":"https://arxiv.org/abs/2503.00031","links":[],"link_count":3,"sections":9},{"id":"cats-conformalized-adaptive-tts-2026","title":"CATS: Conformalized Adaptive Test-Time Scaling","year":2026,"venue":"CAO Workshop at ICLR 2026 Oral","authors":["Mohammad Sadegh Akhondzadeh","Soroush H. Zargarbashi","Simone Antonelli","Aleksandar Bojchevski"],"authors_zh":"Mohammad Sadegh Akhondzadeh、Aleksandar Bojchevski（科隆大学）；Soroush H. Zargarbashi、Simone Antonelli（CISPA 亥姆霍兹信息安全中心）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","conformal-prediction","risk-control","adaptive-thinking","calibration"],"status":"verified","priority":"可读","paper_type_zh":"带保形风险控制的自适应测试时扩展研究","best_for_zh":"需要在推理质量风险与延迟之间设置明确约束的读者。","confidence":"high","one_line":["CATS selects the least costly reasoning-effort level that satisfies a conformal bound on the probability of an incorrect answer.","CATS 在给定错误风险上限下，为每个问题自动选择成本最低的推理强度。"],"why":"It turns adaptive thinking into a risk-controlled decision rather than an unconstrained heuristic for saving compute.","primary_link":"https://openreview.net/forum?id=mXuUomGc0I","links":[],"link_count":1,"sections":9},{"id":"carr-citation-aware-rubric-rewards-2026","title":"Chaining the Evidence: Robust Reinforcement Learning for Deep Search Agents with Citation-Aware Rubric Rewards","year":2026,"venue":"ACL 2026","authors":["Jiajie Zhang","Xin Lv","Ling Feng","Lei Hou","Juanzi Li"],"authors_zh":"Jiajie Zhang, Xin Lv, Ling Feng, Lei Hou, Juanzi Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["deep-research","citation-grounding","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"面向深度检索智能体的引用感知细粒度奖励与数据发布论文","best_for_zh":"需要把深度检索的证据、引用与答案链路转成可训练细粒度奖励信号的研究者。","confidence":"high","one_line":["CaRR turns deep-search answers into verifiable entity, citation, and evidence-chain rewards instead of a single outcome signal.","将复杂检索问题拆成可验证单跳 rubric，并以实体、引用正确性和证据链完整度提供细粒度奖励。"],"why":"It makes factual grounding and evidence connectivity directly optimizable.","primary_link":"https://aclanthology.org/2026.acl-long.950/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/CaRR"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THU-KEG/CaRR-DeepDive"}],"link_count":5,"sections":9},{"id":"challenging-the-boundaries-of-reasoning-an-olympiad-level-math-benchmark-for-large-langu","title":"Challenging the Boundaries of Reasoning: An Olympiad-Level Math Benchmark for Large Language Models","year":2026,"venue":"arXiv","authors":["Haoxiang Sun","Yingqian Min","Zhipeng Chen","Wayne Xin Zhao","Ji-Rong Wen"],"authors_zh":"Haoxiang Sun, Yingqian Min, Zhipeng Chen, Wayne Xin Zhao, Ji-Rong Wen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Three hundred fifty bilingual olympiad problems unify numeric and Lean verification and release 582K+ trajectories.","350 道双语奥赛题统一数值答案和 Lean 两类验证，并开放 582K+ 轨迹。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://aclanthology.org/2026.acl-long.792/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCAIBox/OlymMATH"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/RUC-AIBOX/OlymMATH"}],"link_count":5,"sections":9},{"id":"complex-reasoning-trm-preference-2026","title":"Characterizing, Evaluating, and Optimizing Complex Reasoning","year":2026,"venue":"ICML 2026 Oral","authors":["Haoran Zhang","Yafu Li","Shi Wang","Shilin Wang","Shunkai Zhang","Xiaoye Qu","Lu Cheng"],"authors_zh":"Haoran Zhang、Yafu Li、Shi Wang、Shilin Wang、Shunkai Zhang、Xiaoye Qu、Lu Cheng","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"复杂推理特征、评测与偏好优化论文","best_for_zh":"研究偏好数据、奖励建模或对齐的读者。","confidence":"medium","one_line":["The paper provides preference or reward feedback data for alignment research.","该论文通过偏好数据刻画、评测并优化复杂推理能力。"],"why":"It makes a feedback object available for alignment training, evaluation, or analysis.","primary_link":"https://arxiv.org/abs/2602.08498","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zzzhr97/TRM-Preference"}],"link_count":2,"sections":9},{"id":"chartverse-sft-2026","title":"ChartVerse: Scaling Chart Reasoning via Reliable Programmatic Synthesis from Scratch","year":2026,"venue":"arXiv preprint","authors":["Zheng Liu","Honglin Lin","Chonghan Qin","Xiaoyang Wang","Xin Gao","Yu Li","Mengzhang Cai","Yun Zhu","Zhanping Zhong","Qizhi Pei","Zhuoshi Pan","Xiaoran Shang","Bin Cui","Conghui He","Wentao Zhang","Lijun Wu"],"authors_zh":"Zheng Liu、Honglin Lin、Chonghan Qin、Xiaoyang Wang、Xin Gao、Yu Li、Mengzhang Cai、Yun Zhu、Zhanping Zhong、Qizhi Pei、Zhuoshi Pan、Xiaoran Shang、Bin Cui、Conghui He、Wentao Zhang、Lijun Wu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["program-grounded-chart-reasoning-synthesis"],"tags":["instruction-demonstration-rationale","arxiv-2601.13606","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"图表推理监督微调及独立强化学习子集","confidence":"high","one_line":["ChartVerse generates charts from executable programs and releases questions with both code solutions and long reasoning traces tied to known ground truth.","ChartVerse 从可执行制图程序出发，公开 180 万条带代码解法和长推理的可验证图表问答。"],"why":"Web-scraped chart data is hard to scale and verify because the latent table, rendering code, and exact numerical answers are usually missing.","primary_link":"https://arxiv.org/abs/2601.13606","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/opendatalab/ChartVerse-SFT-1.8M"}],"link_count":2,"sections":9},{"id":"chimera-compact-synthetic-data-for-generalizable-llm-reasoning","title":"CHIMERA: Compact Synthetic Data for Generalizable LLM Reasoning","year":2026,"venue":"arXiv","authors":["Xinyu Zhu","Yihao Feng","Yanchao Sun","Xianzhi Du","Pingzhi Li","Olli Saarikivi","Yun Zhu","Yu Meng"],"authors_zh":"Xinyu Zhu, Yihao Feng, Yanchao Sun, Xianzhi Du, Pingzhi Li, Olli Saarikivi, Yun Zhu, Yu Meng","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** CHIMERA uses a hierarchical taxonomy and multi-model cross-validation to build about 9K compact reasoning examples across eight disciplines.","CHIMERA 用层级 taxonomy 和多模型交叉验证构造约 9K 条覆盖八学科的紧凑推理数据。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2603.00889","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TianHongZXY/CHIMERA"}],"link_count":2,"sections":9},{"id":"chronos-temporal-tts-2026","title":"Chronos: Learning Temporal Dynamics of Reasoning Chains for Test-Time Scaling","year":2026,"venue":"Findings of ACL 2026","authors":["Kai Zhang","Jiayi Liao","Chengpeng Li","Ziyuan Xie","Sihang Li","Xiang Wang"],"authors_zh":"Kai Zhang、Jiayi Liao、Chengpeng Li、Ziyuan Xie、Sihang Li、Xiang Wang（机构：中国科学技术大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-compute","trajectory-scoring","weighted-voting","temporal-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"轨迹评分与并行测试时扩展研究（Findings of ACL 2026）","best_for_zh":"希望改进候选聚合、但不增加大型过程奖励模型的读者。","confidence":"high","one_line":["Chronos scores the temporal evolution of token confidence in sampled reasoning paths before weighted answer aggregation.","Chronos 在加权聚合前评估采样推理路径中词元置信度的时间演化。"],"why":"It turns unequal trajectory quality into an explicit test-time weighting decision rather than treating every sampled answer equally.","primary_link":"https://aclanthology.org/2026.findings-acl.1376/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Hizkai/Chronos"}],"link_count":3,"sections":9},{"id":"longmab-bandit-preference-data-2026","title":"Chunks as Arms: Multi-Armed Bandit-Guided Sampling for Long-Context LLM Preference Optimization","year":2026,"venue":"ACL 2026","authors":["Shaohua Duan","Pengcheng Huang","Xinze Li","Zhenghao Liu","Xiaoyuan Yi","Yukun Yan","Shuo Wang","Yu Gu","Ge Yu","Maosong Sun"],"authors_zh":"Shaohua Duan, Pengcheng Huang, Xinze Li, Zhenghao Liu, Xiaoyuan Yi, Yukun Yan, Shuo Wang, Yu Gu, Ge Yu, Maosong Sun","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["long_context","question_answering","reasoning"],"tags":["preference-data","long-context","multi-armed-bandit","dpo","rollout-selection"],"status":"verified","priority":"必读","paper_type_zh":"长上下文偏好数据构造研究","best_for_zh":"适合构造可程序验证的长上下文问答偏好数据，并研究证据选择如何改变回答对分布的读者。","confidence":"high","one_line":["LongMab uses bandit-guided context chunk rollouts to produce diverse, reward-ranked long-context response pairs for DPO.","LongMab 通过老虎机引导的上下文分块 rollout，为 DPO 生成多样且按奖励排序的长上下文回答对。"],"why":"It makes the context subset that generated an answer part of the preference-data construction process, not merely an inference-time retrieval choice.","primary_link":"https://aclanthology.org/2026.acl-long.161/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NEUIR/LongMab-PO"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/NEUIR/LongMab-PO"}],"link_count":5,"sections":9},{"id":"cl-bench-context-learning-2026","title":"CL-bench: A Benchmark for Context Learning","year":2026,"venue":"arXiv","authors":["Shihan Dou","Ming Zhang","Zhangyue Yin","Chenhao Huang","Yujiong Shen","Junzhe Wang","Jiayi Chen","Yuchen Ni","Junjie Ye","Cheng Zhang","Huaibing Xie","Jianglu Hu","Shaolei Wang","Weichao Wang","Yanling Xiao","Yiting Liu","Zenan Xu","Zhen Guo","Pluto Zhou","Tao Gui","Zuxuan Wu","Xipeng Qiu","Qi Zhang","Xuanjing Huang","Yu-Gang Jiang","Di Wang","Shunyu Yao"],"authors_zh":"Shihan Dou, Ming Zhang, Zhangyue Yin 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["cross_domain","long_context"],"tags":["context_learning","long_context","rubric"],"status":"verified","priority":"可读","paper_type_zh":"上下文学习的专家 rubric 评测基准","best_for_zh":"研究长上下文代理、上下文学习与可验证评测的读者。","confidence":"high","one_line":["CL-bench tests context learning through 1,899 expert-rubriced tasks across 500 complex professional contexts.","CL-bench 以 500 个复杂专业上下文、1,899 项任务和 31,607 条专家 rubric，衡量模型是否能从上下文习得新知识。"],"why":"It shifts long-context evaluation from retrieval toward verifiable acquisition and use of new knowledge.","primary_link":"https://arxiv.org/abs/2602.03587","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent-Hunyuan/CL-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tencent/CL-bench"}],"link_count":4,"sections":9},{"id":"claim-verification-llm-survey-2026","title":"Claim Verification in the Age of Large Language Models: A Survey","year":2026,"venue":"ACL 2026 Student Research Workshop","authors":["Alphaeus Dmonte","Roland R Oruche","Marcos Zampieri","Prasad Calyam","Isabelle Augenstein"],"authors_zh":"Alphaeus Dmonte 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["claim-verification","retrieval","factuality"],"tags":["foundations-and-primers","claim-verification","acl-2026","survey"],"status":"verified","priority":"可读","paper_type_zh":"声明核验综述","best_for_zh":"关注事实性、检索增强生成和自动核验的读者。","confidence":"high","one_line":["An ACL 2026 survey of LLM claim-verification pipelines, including retrieval, prompting, fine-tuning, and English datasets.","梳理基于大语言模型的声明核验流程、检索、提示、微调与英文数据集。"],"why":"It separates evidence retrieval from the model's final decision, which prevents a single score from hiding the cause of failure.","primary_link":"https://aclanthology.org/2026.acl-srw.2/","links":[],"link_count":2,"sections":9},{"id":"claimdb-structured-fact-verification-2026","title":"ClaimDB: A Fact Verification Benchmark over Large Structured Data","year":2026,"venue":"ACL 2026","authors":["Michael Theologitis","Preetam Prabhu Srikar Dammu","Chirag Shah","Dan Suciu"],"authors_zh":"Michael Theologitis, Preetam Prabhu Srikar Dammu, Chirag Shah, Dan Suciu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality","structured_data"],"tags":["fact_verification","structured_data","executable_reasoning","benchmark"],"status":"verified","priority":"可读","paper_type_zh":"大规模结构化数据上的可执行事实核验基准","best_for_zh":"需要评估数据库推理、事实核验或证据不足判断的研究者。","confidence":"high","one_line":["ClaimDB turns large-database fact verification into executable reasoning with three-way evidence labels.","ClaimDB 以真实多表数据库和三类证据标签，测试模型能否用可执行程序完成大规模事实核验。"],"why":"It exposes failures that text-only factuality benchmarks cannot reveal.","primary_link":"https://arxiv.org/abs/2601.14698","links":[{"key":"code","label":["Code","代码"],"url":"https://claimdb.github.io"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/michaeltheologitis/claimdb"}],"link_count":4,"sections":9},{"id":"anthropic-claude-opus-4-6-system-card-2026","title":"Claude Opus 4.6 System Card","year":2026,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation","audit","test_time_compute"],"construction_layer":["trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","agentic_tool_use","cybersecurity","mathematics","safety","long_context"],"tags":["anthropic","claude-opus-4-6","claude","frontier-report","system-card","data-disclosure-ledger","rlhf","rlaif","reasoning-transcripts","reasoning workspace","reward-hacking","evaluation-integrity","interpretability"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露台账","best_for_zh":"审查前沿后训练中有界实现披露、奖励例外、评测完整性和 4.6 与 4.5 证据边界的读者","confidence":"high","one_line":["Claude Opus 4.6's system card partially discloses source categories, RLHF/RLAIF, prior-model reasoning-transcript initialization, a <0.01% reasoning workspace-reward error, and detailed safety/audit practice, but not the underlying data, reward, or reproducibility artifacts.","Claude Opus 4.6 的系统卡部分披露了来源类别、RLHF/RLAIF、先前模型 reasoning-transcript 初始化、小于 0.01% 的 reasoning workspace-reward 错误以及细致的安全/审计实践，但没有公开底层数据、奖励或可复现制品。"],"why":"It is a high-value Track 12 disclosure ledger item because it clearly separates a rare, source-specific implementation disclosure from the much larger body of unavailable records, reward specifications, and audit artifacts—and marks where the report merely refers to earlier 4.5 methods.","primary_link":"https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf","links":[],"link_count":2,"sections":9},{"id":"anthropic-claude-sonnet-4-6-system-card-2026","title":"Claude Sonnet 4.6 System Card","year":2026,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["safety_alignment"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","coding","agentic_tool_use"],"tags":["anthropic","claude-sonnet-4-6","system-card","reinforcement-from-ai-feedback","preference-selection","safety-evaluation","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露账本","best_for_zh":"审计闭源前沿模型报告中已披露与未披露的训练数据、反馈、审计和可复现性边界","confidence":"high","one_line":["The Claude Sonnet 4.6 System Card reports broad proprietary training-source categories, deduplication/classification, post-training, reinforcement from AI feedback, and worker preference/safety roles, while record-level data, reward, and recipe details remain undisclosed.","《Claude Sonnet 4.6 System Card》披露了宽泛的专有训练数据类别、去重/分类、后训练、来自 AI 的强化反馈以及人工偏好和安全工作角色，但记录级数据、奖励契约和训练配方仍未公开。"],"why":"It provides an auditable closed-frontier baseline: readers can distinguish the disclosed source categories, high-level feedback methods, and safety-audit practice from the missing data lineage, reward contract, rollout policy, and reproducibility evidence needed to reuse or independently assess a reasoning-data pipeline.","primary_link":"https://www-cdn.anthropic.com/bbd8ef16d70b7a1665f14f306ee88b53f686aa75/Claude%20Sonnet%204.6%20System%20Card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://www.anthropic.com/research/claude-sonnet-4-6"}],"link_count":3,"sections":9},{"id":"clawbench-online-tasks-2026","title":"ClawBench: Can AI Agents Complete Everyday Online Tasks?","year":2026,"venue":"arXiv preprint","authors":["Yuxuan Zhang","Yubo Wang","Yipeng Zhu","Penghui Du","Junwen Miao","Xuan Lu","Wendong Xu","Yunzhuo Hao","Songcheng Cai","Xiaochen Wang","Huaisong Zhang","Xian Wu","Yi Lu","Minyi Lei","Kai Zou","Huifeng Yin","Ping Nie","Liang Chen","Dongfu Jiang","Wenhu Chen","Kelsey R. Allen"],"authors_zh":"Yuxuan Zhang、Yubo Wang、Yipeng Zhu 等","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-agents","online-tasks","production-websites"],"tags":["benchmark","agent_environment","web-agents"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / live web agent environment","best_for_zh":"需要审计 live-web agent、生产网站漂移、账号状态、最终提交拦截和真实副作用边界的读者。","confidence":"medium","one_line":["ClawBench evaluates web agents on everyday online tasks across live production websites.","ClawBench 在 144 个 live platforms 上评估 153 个日常在线任务，并用 final-submission interception 控制真实副作用。"],"why":"It is a live-web evaluation surface where environmental drift and real-world side-effect control are central.","primary_link":"https://arxiv.org/abs/2604.08523","links":[{"key":"project","label":["Project","项目主页"],"url":"https://claw-bench.com"}],"link_count":3,"sections":9},{"id":"oda-mixture-2026","title":"Closing the Data Loop: Using OpenDataArena to Engineer Superior Training Datasets","year":2026,"venue":"arXiv preprint","authors":["Gao, Xin","Wang, Xiaoyang","Zhu, Yun","Cai, Mengzhang","He, Conghui","Wu, Lijun"],"authors_zh":"Gao, Xin、Wang, Xiaoyang、Zhu, Yun、Cai, Mengzhang、He, Conghui、Wu, Lijun","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["evaluation-guided-open-SFT-mixture-engineering"],"tags":["instruction-demonstration-rationale","arxiv-2601.09733","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"通用推理监督微调","confidence":"high","one_line":["ODA closes the loop by ranking source corpora in OpenDataArena, then deduplicating and decontaminating the best-performing 500K-record mixture.","ODA 用公开数据竞技场的来源级反馈选择、去重和去污染约 50 万条通用后训练记录。"],"why":"Open SFT mixtures are often assembled heuristically without feedback linking source-level data choices to downstream capability.","primary_link":"https://arxiv.org/abs/2601.09733","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenDataArena/ODA-Mixture-500k"}],"link_count":2,"sections":9},{"id":"diffcodegen-differential-tts-2026","title":"Code Generation by Differential Test Time Scaling","year":2026,"venue":"arXiv preprint","authors":["Yifeng He","Ethan Wang","Jicheng Wang","Xuanxin Ouyang","Hao Chen"],"authors_zh":"Yifeng He、Ethan Wang、Jicheng Wang、Xuanxin Ouyang、Hao Chen（机构：加州大学戴维斯分校、武汉大学、香港大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["software-engineering","code-generation"],"tags":["test-time-scaling","code-generation","differential-testing","fuzzing","candidate-selection"],"status":"verified","priority":"必读","paper_type_zh":"高效代码生成测试时扩展研究","best_for_zh":"没有公开测试用例、却需要以可执行分析替代奖励模型重排序的代码助手开发者。","confidence":"high","one_line":["DiffCodeGen uses coverage-guided differential testing to select among diverse code candidates with almost no extra LLM-token cost beyond candidate generation.","DiffCodeGen 利用覆盖引导的差分测试从多样代码候选中选择结果，除候选生成外几乎不增加大语言模型词元成本。"],"why":"It turns the behavioral agreement of independently generated programs into an inexpensive verifier and shifts test-time budget from model calls to executable analysis.","primary_link":"https://arxiv.org/abs/2605.20473","links":[],"link_count":1,"sections":9},{"id":"codejudgebench-2025","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","year":2026,"venue":"ACL 2026","authors":["Hongchao Jiang","Yiming Chen","Yushi Cao","Hung-yi Lee","Robby T. Tan"],"authors_zh":"Hongchao Jiang, Yiming Chen, Yushi Cao, Hung-yi Lee, Robby T. Tan","tracks":["audit_failure_contamination_verifier_attacks","benchmarks_evaluation_surfaces","preference_reward_feedback_data","judgment_rubric_domain_expert_data"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"代码 LLM-as-a-Judge 基准与可靠性审计","best_for_zh":"需要比较、校准或审计代码生成、修复和单元测试判别器的研究者。","confidence":"high","one_line":["Coding judge benchmark covering generation, repair, and unit-test judgments and their order sensitivity.","用 5,352 个经验证候选对系统测试代码 LLM 判别器，并暴露顺序和来源敏感性。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://aclanthology.org/2026.acl-long.888/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hongchaoy/CodeJudgeBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mattymchen/codejudgebench"}],"link_count":5,"sections":9},{"id":"coderm-nt-2026","title":"CodeRM-NT: Reward Model for Code RL without Unit Tests","year":2026,"venue":"Findings of ACL 2026","authors":["Xiao Xia","Dan Zhang","Tianrui Sun"],"authors_zh":"Xiao Xia, Dan Zhang, Tianrui Sun","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"代码奖励模型与强化学习方法论文","best_for_zh":"需要在缺少可靠单元测试时构建、审计或复现代码强化学习奖励的研究者。","confidence":"high","one_line":["A code reward-model pipeline that judges execution traces and releases the associated code/data artifact.","CodeRM-NT 以 MCTS 生成的执行轨迹和 LLM judge 训练代码奖励模型，不依赖单元测试即可为代码 RL 提供反馈。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.findings-acl.2150/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/CodeRM-NT"}],"link_count":2,"sections":9},{"id":"codescaler-execution-free-reward-models-code-2026","title":"CodeScaler: Scaling Code LLM Training and Test-Time Inference via Execution-Free Reward Models","year":2026,"venue":"arXiv","authors":["Xiao Zhu","Xinyu Zhou","Boyu Zhu","Hanxu Hu","Mingzhe Du","Haotian Zhang","Huiming Wang","Zhijiang Guo"],"authors_zh":"Xiao Zhu、Xinyu Zhou、Boyu Zhu、Hanxu Hu、Mingzhe Du、Haotian Zhang、Huiming Wang、Zhijiang Guo","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["code-generation","reward-modeling"],"tags":["code","reward-model","preference-pairs","test-time-scaling","2026"],"status":"verified","priority":"可读","paper_type_zh":"代码偏好数据、无执行奖励模型与推理时选择方法","best_for_zh":"需要代码偏好对、代码奖励模型或免执行候选重排方法的研究者","confidence":"high","one_line":["CodeScaler releases 51,107 verified code preference pairs for training execution-free reward models that scale code training and test-time selection.","CodeScalerPair-51K 含 51,107 个经验证的代码候选偏好对，以无执行奖励模型支持代码推理训练与 test-time 选择。"],"why":"It targets the practical setting where executing every candidate program is too costly while retaining pairwise reward supervision.","primary_link":"https://arxiv.org/abs/2602.17684","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LARK-AI-Lab/CodeScaler"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LARK-Lab/CodeScalerPair-51K"},{"key":"project","label":["Project","项目主页"],"url":"https://lark-ai-lab.github.io/codescaler.github.io/"}],"link_count":5,"sections":9},{"id":"codescout-code-search-rollouts-2026","title":"CodeScout: An Effective Recipe for Reinforcement Learning of Code Search Agents","year":2026,"venue":"ACM CAIS 2026 Workshop on Agentic Software Engineering (non-archival)","authors":["Lintang Sutawika","Aditya Bharat Soni","Bharath Sriraam R R","Apurva Gandhi","Taha Yassine","Sanidhya Vijayvargiya","Yuchen Li","Xuhui Zhou","Yilin Zhang","Leander Melroy Maben","Graham Neubig"],"authors_zh":"Lintang Sutawika、Aditya Bharat Soni、Bharath Sriraam R R、Apurva Gandhi、Taha Yassine、Sanidhya Vijayvargiya、Yuchen Li、Xuhui Zhou、Yilin Zhang、Leander Melroy Maben、Graham Neubig","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training","sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["software_engineering","repository_level_code","code_localization","agent_tool_use"],"tags":["codescout","code-localization","software-engineering-agents","openhands","terminal-tool-use","multi-attempt-rollouts","zero-reward-failures","gspo","rlvr","patch-derived-verifier","full-episode-traces"],"status":"partial","priority":"必读","paper_type_zh":"代码搜索 agent 的在线 RL rollout 数据发布与可验证奖励构造配方","best_for_zh":"研究多次 rollout、代码定位 verifier、失败轨迹保留、异步 RLVR、训练预算核对与 agent 轨迹可重放性的读者","confidence":"high","one_line":["Releases 54,845 full terminal-search RL rollouts for CodeScout-14B and CodeScout-4B, preserving grouped attempts, component localization rewards, and 8,281 zero-reward failures rather than publishing only successful trajectories.","CodeScout 发布 54,845 条 CodeScout-14B 与 CodeScout-4B 的完整终端代码搜索 RL rollout，保留多次尝试、分层定位奖励和 8,281 条零奖励失败轨迹；但公开行数超过论文名义预算，部分分组不完整，且数据许可与精确重放谱系仍为 unknown。"],"why":"It exposes how repository-search attempts, a patch-derived terminal verifier, multiple rollouts, and asynchronous GSPO interact, while the count mismatch, incomplete groups, missing replay lineage, and absent data license show exactly what an auditable agent-trajectory release still needs.","primary_link":"https://arxiv.org/abs/2603.17829","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenHands/codescout"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenHands/CodeScout_Training_Rollouts"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/OpenHands/codescout"}],"link_count":8,"sections":9},{"id":"codetracer-traceable-agent-states-2026","title":"CodeTracer: Towards Traceable Agent States","year":2026,"venue":"arXiv preprint","authors":["Han Li","Yifan Yao","Letian Zhu","Rili Feng","Hongyi Ye","Jiaming Wang","Yancheng He","Pengyu Zou","Lehan Zhang","Xinping Lei","Haoyang Huang","Ken Deng","Ming Sun","Zhaoxiang Zhang","He Ye","Jiaheng Liu"],"authors_zh":"Han Li, Yifan Yao, Letian Zhu, Rili Feng, Hongyi Ye, Jiaming Wang, Yancheng He, Pengyu Zou, Lehan Zhang, Xinping Lei, Haoyang Huang, Ken Deng, Ming Sun, Zhaoxiang Zhang, He Ye, Jiaheng Liu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["code-agents","process-supervision","failure-localization"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要定位代码代理失败起点和下游错误链的研究者。","confidence":"high","one_line":["CodeTraceBench releases code-agent traces with stage- and step-level failure labels for diagnosis and recovery.","CodeTraceBench 提供代码代理轨迹及阶段、步骤级失败标签，用于过程诊断、奖励建模与恢复策略。"],"why":"The release provides a reusable process-supervision or process-evaluation surface.","primary_link":"https://arxiv.org/abs/2604.11641","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NJU-LINK/CodeTracer"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/NJU-LINK/CodeTraceBench"}],"link_count":3,"sections":9},{"id":"coig-p-chinese-preference-2025","title":"COIG-P: A High-Quality and Large-Scale Chinese Preference Dataset for Alignment with Human Values","year":2026,"venue":"Findings of EACL 2026","authors":["Siwei Wu","Jincheng Ren","Xinrun Du","Shuyue Guo","Xingwei Qu","Yiming Liang","Jie Liu","Yunwen Li","Tyler Loakman","Tianyu Zheng","Boyu Feng","Huaqing Yuan","Zenith Wang","Jiaheng Liu","Wenhao Huang","Chenglin Cai","Haoran Que","Jian Yang","Yuelin Bai","Zekun Moore Wang","Zhouliang Yu","Qunshu Lin","Ding Pan","Yuchen Jiang","Tiannan Wang","Wangchunshu Zhou","Shenzhi Wang","Xingyuan Bu","Minghao Liu","Guoyin Wang","Ge Zhang","Chenghua Lin"],"authors_zh":"Siwei Wu、Jincheng Ren、Xinrun Du、Shuyue Guo、Xingwei Qu、Yiming Liang、Jie Liu、Yunwen Li、Tyler Loakman、Tianyu Zheng、Boyu Feng、Huaqing Yuan、Zenith Wang、Jiaheng Liu、Wenhao Huang、Chenglin Cai、Haoran Que、Jian Yang、Yuelin Bai、Zekun Moore Wang、Zhouliang Yu、Qunshu Lin、Ding Pan、Yuchen Jiang、Tiannan Wang、Wangchunshu Zhou、Shenzhi Wang、Xingyuan Bu、Minghao Liu、Guoyin Wang、Ge Zhang、Chenghua Lin","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["COIG-P releases Chinese chosen/rejected pairs across six task domains, using multi-model generation and judging to make preference optimization resources available at million-pair scale.","COIG-P 发布覆盖六类任务的中文 chosen/rejected 对，通过多模型生成与评审提供百万级偏好优化资源。"],"why":"About one million Chinese chosen/rejected pairs across six domains, generated and judged by multiple LLMs.","primary_link":"https://aclanthology.org/2026.findings-eacl.288/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/multimodal-art-projection/COIG-P"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/m-a-p/COIG-P"}],"link_count":3,"sections":9},{"id":"complex-if-expert-rubrics-rlvr-2026","title":"Complex-IF and Beyond: Expert Rubrics for RLVR","year":2026,"venue":"GEM 2026 Workshop","authors":["Sushant Mehta","Liudas Panavas","Eleanor Fleming","Paul Mains","Edwin Chen"],"authors_zh":"Sushant Mehta、Liudas Panavas、Eleanor Fleming、Paul Mains、Edwin Chen","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","sft","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["llm_judge","rubric_evaluation"],"tags":["rubric","llm_judge","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"专家 rubric 数据与训练论文","best_for_zh":"用专家准则构建复杂指令评测或 RLVR 奖励的研究者。","confidence":"high","one_line":["Complex-IF pairs difficult instructions with expert atomic rubrics to support both evaluation and rubric-reward RLVR.","将约千条复杂指令与 10–40 个专家原子 rubric 结合，作为评测和 RLVR 的稠密奖励信号。"],"why":"It exposes rubric-grounded evaluation signals and their audit boundary.","primary_link":"https://aclanthology.org/2026.gem-main.61/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/surgeai/ComplexConstraints"}],"link_count":3,"sections":9},{"id":"complexconstraints-rlvr-2026","title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","year":2026,"venue":"GEM Workshop at ACL 2026","authors":["Sushant Mehta","Liudas Panavas","Suhaas Garre","Edwin Chen"],"authors_zh":"Sushant Mehta, Liudas Panavas, Suhaas Garre, Edwin Chen","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["instruction_following","agent_workflows"],"tags":["rubric","rlvr","instruction_following"],"status":"verified","priority":"可读","paper_type_zh":"专家 rubric 驱动的 RLVR 基准与训练数据","best_for_zh":"需要研究指令跟随奖励、rubric Judge 或 RLVR 的研究者。","confidence":"high","one_line":["ComplexConstraints shows expert atomic rubrics can train and evaluate complex instruction following.","ComplexConstraints 以专家原子化 rubric 为复杂指令跟随和 Agent 强化学习提供可解释奖励。"],"why":"It connects interpretable rubric judgments to RL across static and stateful tasks.","primary_link":"https://arxiv.org/abs/2606.09118","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/surgeai/ComplexConstraints"}],"link_count":4,"sections":9},{"id":"complex-mcp-2026","title":"ComplexMCP: Evaluation of LLM Agents in Dynamic, Interdependent, and Large-Scale Tool Sandbox","year":2026,"venue":"ICML 2026","authors":["Yuanyang Li","Xue Yang","Longyue Wang","Weihua Luo","Hongyang Chen"],"authors_zh":"Yuanyang Li, Xue Yang, Longyue Wang, Weihua Luo, Hongyang Chen","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","software_automation","mcp"],"tags":["environment-agent-trajectory-data","agent-benchmark","mcp","tool-use","stateful-environment","programmatic-verification","gold-trajectories","prompt-injection-risk","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"可执行 MCP 智能体基准与成功参考轨迹发布","best_for_zh":"研究环境智能体轨迹、状态差分 verifier、工具检索与可复现评测审计的读者","confidence":"high","one_line":["ComplexMCP pairs 47 manually curated tasks with seeded 315-tool environments, accepted gold tool trajectories, and final-state verification, while leaving full model rollouts and a paper-era release freeze unavailable.","ComplexMCP 将 47 个手工任务、15 个 MCP server 的 315 个工具、成功参考轨迹与状态差分 verifier 组合为可执行评测对象，但未发布完整模型 rollout、论文时期版本锁与独立 rights 声明。"],"why":"It makes long-horizon agent reasoning inspectable at the state-action level—queries, tool calls, observations, target state, collateral changes, and success—but also shows why executable benchmarks need immutable releases, verifier tests, security calibration, and complete rollout retention before reuse as post-training data.","primary_link":"https://openreview.net/pdf?id=3u1M9wokD7","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ATH-MaaS/complex-mcp"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ATH-MaaS/complex-mcp/blob/617e963bd838bee5793a39e6b34165b79535828f/benchmark/data/data.parquet"}],"link_count":6,"sections":9},{"id":"cogs-composition-grounded-instruction-synthesis-2026","title":"Composition-Grounded Instruction Synthesis for Visual Reasoning","year":2026,"venue":"ICLR 2026","authors":["Xinyi Gu","Jiayuan Mao","Zhang-Wei Hong","Zhuoran Yu","Pengyuan Li","Dhiraj Joshi","Rogerio Feris","Zexue He"],"authors_zh":"Xinyi Gu、Jiayuan Mao、Zhang-Wei Hong、Zhuoran Yu、Pengyuan Li、Dhiraj Joshi、Rogerio Feris、Zexue He","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["visual-reasoning","synthetic-data","process-reward"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要合成图表、文档或网页视觉推理数据，并让中间子问题直接形成过程奖励的研究者。","confidence":"high","one_line":["COGS decomposes a few difficult visual questions into reusable perception and reasoning factors, then recomposes them with new images to synthesize QA data with factor-level process rewards.","COGS 将少量困难视觉问题分解为可复用的感知与推理因子，再与新图像重组，合成带因子级过程奖励的问答数据。"],"why":"It releases a practical path from scarce seed supervision to image-grounded intermediate-answer supervision without manually labeling every trajectory.","primary_link":"https://arxiv.org/abs/2510.15040","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/gxy000/COGS"},{"key":"data","label":["Data","数据"],"url":"https://www.dropbox.com/scl/fo/db29pn3bya6n6l3pzs9yo/ALu7ecV3LgH-gcvfyESW8vQ/data?dl=0&rlkey=fl4ty5oovx8xvo48k2dhx58q1&st=njr870c9&subfolder_nav_tracking=1"},{"key":"project","label":["Project","项目主页"],"url":"https://cogsynthesis.github.io/"}],"link_count":5,"sections":9},{"id":"composition-rl-compose-your-verifiable-prompts-for-rl-of-llms","title":"Composition-RL: Compose Your Verifiable Prompts for Reinforcement Learning of Large Language Models","year":2026,"venue":"arXiv","authors":["Xin Xu","Clive Bai","Kai Yang","Tianhao Chen","Yangkun Chen","Weijie Liu","Hao Chen","Yang Wang","Saiyong Yang","Can Yang"],"authors_zh":"Xin Xu, Clive Bai, Kai Yang, Tianhao Chen, Yangkun Chen, Weijie Liu, Hao Chen, Yang Wang, Saiyong Yang, Can Yang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Composition-RL turns saturated easy verifiable prompts into 1.323M harder composed tasks that restore RL signal.","Composition-RL 将已饱和的可验证简单题组合成 132.3 万更难 prompt，继续提供 RL 信号。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2602.12036","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XinXU-USTC/Composition-RL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xx18/Polaris-Composition-1323K"}],"link_count":3,"sections":9},{"id":"congrad-multilingual-preference-filtering-2026","title":"CONGRAD: Conflicting Gradient Filtering for Multilingual Preference Alignment","year":2026,"venue":"EACL 2026 Long Papers","authors":["Jiangnan Li","Thuy-Trang Vu","Christian Herold","Amirhossein Tebbifakhr","Shahram Khadivi","Gholamreza Haffari"],"authors_zh":"Jiangnan Li, Thuy-Trang Vu, Christian Herold, Amirhossein Tebbifakhr, Shahram Khadivi, Gholamreza Haffari（莫纳什大学、eBay Inc.）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","multilingual_learning","preference_learning"],"tags":["multilingual-alignment","preference-data","gradient-filtering","self-rewarding","dpo"],"status":"verified","priority":"必读","paper_type_zh":"多语言偏好数据筛选与迭代对齐研究","best_for_zh":"适合设计多语言偏好学习流程、希望保留跨语言迁移又避免一种语言的更新损害另一种语言的读者。","confidence":"high","one_line":["CONGRAD filters self-generated multilingual preference pairs by their alignment with a de-conflicted consensus gradient before DPO consumes them.","CONGRAD 在 DPO 使用自生成的多语言偏好回答对之前，依据其与去冲突共识梯度的一致程度进行筛选。"],"why":"It operationalizes cross-language training interference as a record-selection criterion instead of treating all generated preference pairs as equally safe to optimize on.","primary_link":"https://aclanthology.org/2026.eacl-long.299/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/KagamiBaka/CONGRAD"}],"link_count":3,"sections":9},{"id":"continual-safety-gradient-selection-2026","title":"Continual Safety Alignment via Gradient-Based Sample Selection","year":2026,"venue":"Findings of ACL 2026","authors":["Thong Bach","Dung Nguyen","Thao Minh Le","Truyen Tran"],"authors_zh":"Thong Bach, Dung Nguyen, Thao Minh Le, Truyen Tran","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","safety"],"tags":["data-selection","continual-fine-tuning","safety-alignment","gradients"],"status":"verified","priority":"可读","paper_type_zh":"基于梯度的持续微调数据选择研究","best_for_zh":"适合把已对齐模型迁移到良性领域、又不希望额外加入安全回放集的读者。","confidence":"high","one_line":["The paper preserves continual safety alignment by excluding high-gradient task records and training on moderate-gradient examples instead.","该研究通过排除高梯度任务记录、保留中等梯度样本，在持续微调中缓解安全对齐漂移。"],"why":"It shows that even benign records have unequal alignment cost and makes that cost a direct training-selection criterion.","primary_link":"https://aclanthology.org/2026.findings-acl.942/","links":[],"link_count":2,"sections":9},{"id":"counsel-meta-evaluation-agentic-tasks-2026","title":"Counsel: A Meta-Evaluation Dataset for Agentic Tasks","year":2026,"venue":"arXiv preprint","authors":["Sashank Pisupati","Henry Broomfield","Eujeong Choi","Antonia Calvi","Charlie Wang","Roman Engeler","Max Bartolo","Patrick Lewis"],"authors_zh":"Sashank Pisupati、Henry Broomfield、Eujeong Choi、Antonia Calvi、Charlie Wang、Roman Engeler、Max Bartolo、Patrick Lewis","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent-evaluation","meta-evaluation","process-critique"],"tags":["process-supervision","agent-trajectories","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要校准或训练智能体过程批评器，并分别评估错误位置与批评理由质量的研究者。","confidence":"high","one_line":["Counsel meta-evaluates process-level agent critiques with human labels that separate correct locations, sound reasoning, and false alarms.","Counsel 用人工元评测标签区分 agent judge 批评的定位正确性、推理质量与误报，为过程批评器校准提供监督。"],"why":"It supervises whether a process critique is justified, rather than treating the judge's own flag as ground truth.","primary_link":"https://arxiv.org/abs/2606.21627","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AtlaAI/counsel"}],"link_count":2,"sections":9},{"id":"credit-budgeted-icpc-style-coding-2026","title":"Credit-Budgeted ICPC-Style Coding: When Agents Must Pay for Every Decision","year":2026,"venue":"ICLR 2026 / arXiv","authors":["Lingfeng Zhou","Junhao Shi","Jin Gao","Dequan Wang"],"authors_zh":"Lingfeng Zhou 等（Shanghai Jiao Tong University、Shanghai Innovation Institute）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["competitive-programming","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","competitive-programming"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2026 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"low","one_line":["Credit-Budgeted ICPC-Style Coding exposes ICPC-style coding with credit-budgeted evaluation as an auditable evaluation surface.","该工作把带额度预算的 ICPC 风格编程评测做成可审计的评测面，让智能体为每一次决策付出成本。"],"why":"Old entry reset; potentially high relevance if source verifies.","primary_link":"https://arxiv.org/abs/2604.10182","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/maple-zhou/USACOArena"}],"link_count":3,"sections":9},{"id":"crowdselect-synthetic-instruction-2026","title":"CROWD SELECT: Synthetic Instruction Data Selection with Multi-LLM Wisdom","year":2026,"venue":"Findings of EACL 2026","authors":["Yisen Li","Lingfeng Yang","Wenxuan Shen","Pan Zhou","Yao Wan","Weiwei Lin","Dongping Chen"],"authors_zh":"Yisen Li, Lingfeng Yang, Wenxuan Shen, Pan Zhou, Yao Wan, Weiwei Lin, Dongping Chen","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"必读","paper_type_zh":"训练数据选择或奖励设计论文","best_for_zh":"研究后训练数据消费与优化目标的读者。","confidence":"high","one_line":["CrowdSelect selects synthetic instruction pairs with complementary signals from multiple LLM responses and reward-model assessments.","CrowdSelect 利用多个语言模型回复和奖励模型评估提供的互补信号选择合成指令对。"],"why":"It makes the training-data decision inspectable.","primary_link":"https://aclanthology.org/2026.findings-eacl.79/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/listentm/crowdselect"}],"link_count":2,"sections":9},{"id":"cua-suite-human-video-demonstrations-2026","title":"CUA-Suite: Massive Human-annotated Video Demonstrations for Computer-Use Agents","year":2026,"venue":"ICLR 2026 Workshop on Lifelong Agents: Learning, Aligning, Evolving","authors":["Xiangru Jian","Shravan Nayak","Kevin Qinghong Lin","Aarash Feizi","Kaixin Li","Patrice Bechard","Spandana Gella","Sai Rajeswar"],"authors_zh":"Xiangru Jian、Shravan Nayak、Kevin Qinghong Lin、Aarash Feizi、Kaixin Li、Patrice Bechard、Spandana Gella、Sai Rajeswar","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["computer-use-agents","gui-trajectories","video-based-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要把连续桌面视频和操作日志转成智能体轨迹，用于模仿学习、离线强化学习或基于视频的奖励建模的研究者。","confidence":"high","one_line":["CUA-Suite releases about 10,000 expert desktop videos with timestamped actions and dense UI links, preserving full interaction trajectories for computer-use training and process evaluation.","CUA-Suite 发布约 1 万条专家桌面操作视频、时间戳动作和稠密 UI 链接，保留完整交互轨迹以支持计算机使用训练与过程评测。"],"why":"It preserves temporal interaction evidence that sparse screenshots and final clicks discard.","primary_link":"https://openreview.net/forum?id=IgTUGrZfMr","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ServiceNow/GroundCUA/tree/main/VideoCUA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ServiceNow/VideoCUA"},{"key":"project","label":["Project","项目主页"],"url":"https://cua-suite.github.io/"}],"link_count":5,"sections":9},{"id":"rubric-rewards-curing-miracle-steps-2026","title":"Curing “Miracle Steps” in LLM Mathematical Reasoning with Rubric Rewards","year":2026,"venue":"ACL 2026","authors":["Youliang Yuan","Qiuyang Mang","Jingbang Chen","Hong Wan","Xiaoyuan Liu","Junjielong Xu","Jen-tse Huang","Wenxuan Wang","Wenxiang Jiao","Pinjia He"],"authors_zh":"Youliang Yuan, Qiuyang Mang, Jingbang Chen, Hong Wan, Xiaoyuan Liu, Junjielong Xu, Jen-tse Huang, Wenxuan Wang, Wenxiang Jiao, Pinjia He","tracks":["preference_reward_feedback_data","process_trace_supervision_data","audit_failure_contamination_verifier_attacks"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","rubric-reward-modeling","reinforcement-learning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"问题特定评分准则与轨迹级奖励建模论文","best_for_zh":"需要用问题特定评分准则诊断推理投机、训练生成式奖励模型或开展数学强化学习的研究者。","confidence":"high","one_line":["Rubric Rewards pairs mathematical problems with problem-specific scoring rubrics so a generative reward model can penalize trajectory-level logical shortcuts during RL.","14.3K 数学题—rubric 样本，以问题级准则评估整条推理轨迹并作 RL 奖励，直接针对推理投机。"],"why":"It makes the feedback contract explicit at the whole-trajectory level when final-answer correctness is insufficient to detect reasoning shortcuts.","primary_link":"https://aclanthology.org/2026.acl-long.844/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YouliangYuan/rrm-cure-miracle-steps"}],"link_count":4,"sections":9},{"id":"d3-gym-constructing-real-world-verifiable-environments-for-data-driven-discovery","title":"D3-Gym: Constructing Real-World Verifiable Environments for Data-Driven Discovery","year":2026,"venue":"arXiv","authors":["Hanane Nour Moussa","Yifei Li","Zhuoyang Li","Yankai Yang","Cheng Tang","Tianshu Zhang","Nesreen K. Ahmed","Ali Payani","Ziru Chen","Huan Sun"],"authors_zh":"Hanane Nour Moussa, Yifei Li, Zhuoyang Li, Yankai Yang, Cheng Tang, Tianshu Zhang, Nesreen K. Ahmed, Ali Payani, Ziru Chen, Huan Sun","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** 565 real scientific discovery tasks binding Docker, data, reference implementations, and automatic evaluators.","565 个真实科学数据发现任务，绑定 Docker、输入数据、参考实现和自动 evaluator。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2604.27977","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OSU-NLP-Group/D3-Gym"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/osunlp/D3-Gym"}],"link_count":3,"sections":9},{"id":"finegrained-preference-data-selection-2026","title":"Data Selection for LLM Alignment Using Fine-Grained Preferences","year":2026,"venue":"ICLR 2026","authors":["Jia Zhang","Yao Liu","Chen-Xi Zhang","Yi Liu","Yi-Xuan Jin","Lan-Zhe Guo","Yu-Feng Li"],"authors_zh":"Jia Zhang、Yao Liu、Chen-Xi Zhang、Yi Liu、Yi-Xuan Jin、Lan-Zhe Guo、Yu-Feng Li（南京大学、阿里巴巴淘天集团）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-following"],"tags":["fine-grained-preferences","data-selection","preference-divergence","direct-preference-optimization"],"status":"verified","priority":"必读","paper_type_zh":"细粒度偏好数据选择与对齐研究","best_for_zh":"适合清洗含多个已标注价值或行为维度的偏好数据的读者。","confidence":"high","one_line":["The method filters fine-grained preference records by cross-aspect consensus before standard DPO, retaining samples with the most negative preference divergence.","该方法先按跨维度偏好共识过滤细粒度偏好记录，再进行标准 DPO，保留偏好分歧最负的样本。"],"why":"It turns cross-aspect conflict into a measurable training-data exclusion rule instead of asking a single holistic label to resolve every value tradeoff.","primary_link":"https://openreview.net/forum?id=FqFU3IrnEa","links":[],"link_count":3,"sections":9},{"id":"mds-multiturn-dialogue-selection-2026","title":"Data Selection for Multi-turn Dialogue Instruction Tuning","year":2026,"venue":"Findings of ACL 2026","authors":["Bo Li","Shikun Zhang","Wei Ye"],"authors_zh":"Bo Li, Shikun Zhang, Wei Ye","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["dialogue","instruction-tuning","customer-service"],"tags":["multi-turn-dialogue","data-selection","sft","coverage"],"status":"verified","priority":"必读","paper_type_zh":"多轮对话指令数据选择研究","best_for_zh":"为对话监督微调筛选连贯且多样的会话历史的读者。","confidence":"high","one_line":["MDS selects complete dialogue trajectories by combining user-intent coverage with entity-grounded and form-consistent structure.","MDS 以用户意图覆盖以及实体扎根、形式一致的结构选择完整对话轨迹。"],"why":"It moves the SFT admission decision from isolated turns to the complete dialogue trajectory.","primary_link":"https://aclanthology.org/2026.findings-acl.130/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WisdomShell/MDS"},{"key":"project","label":["Project","项目主页"],"url":"https://wisdomshell.github.io/MDS/"}],"link_count":5,"sections":9},{"id":"cropi-offpolicy-influence-rlvr-2026","title":"Data-Efficient RLVR via Off-Policy Influence Guidance","year":2026,"venue":"ACL 2026","authors":["Erle Zhu","Dazhi Jiang","Yuan Wang","Xujun Li","Jiale Cheng","Yuxian Gu","Yilin Niu","Aohan Zeng","Jie Tang","Minlie Huang","Hongning Wang"],"authors_zh":"Erle Zhu, Dazhi Jiang, Yuan Wang, Xujun Li, Jiale Cheng, Yuxian Gu, Yilin Niu, Aohan Zeng, Jie Tang, Minlie Huang, Hongning Wang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics"],"tags":["rlvr","influence-functions","off-policy","curriculum","data-selection"],"status":"verified","priority":"必读","paper_type_zh":"离策略影响引导的 RLVR 数据选择研究","best_for_zh":"适合构建目标导向或算力受限的 RLVR 系统，并希望依据当前策略选择数据、却不愿为全提示池重新生成轨迹的读者。","confidence":"high","one_line":["CROPI uses offline trajectories and projected gradient influence to refresh the 10-percent RLVR subset that receives each stage of GRPO training.","CROPI 用离线轨迹和投影梯度影响分数，持续更新每个阶段实际进入 GRPO 训练的 10% RLVR 提示。"],"why":"It connects a prompt-selection decision to an explicit approximation of its expected contribution to validation objectives, then refreshes that decision during training.","primary_link":"https://aclanthology.org/2026.acl-long.2141/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thu-coai/CROPI"}],"link_count":3,"sections":9},{"id":"davinci-env-open-swe-environment-synthesis-at-scale","title":"daVinci-Env: Open SWE Environment Synthesis at Scale","year":2026,"venue":"arXiv","authors":["Dayuan Fu","Shenyu Wu","Yunze Wu","Zerui Peng","Yaxing Huang","Jie Sun","Ji Zeng","Mohan Jiang","Lin Zhang","Yukun Li","Jiarui Hu","Liming Liu","Jinlong Hou","Pengfei Liu"],"authors_zh":"Dayuan Fu, Shenyu Wu, Yunze Wu, Zerui Peng, Yaxing Huang, Jie Sun, Ji Zeng, Mohan Jiang, Lin Zhang, Yukun Li, Jiarui Hu, Liming Liu, Jinlong Hou, Pengfei Liu","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** daVinci-Env turns more than 12.8K repositories into 45,320 environments and collects 13K trajectories from about 9K of them.","daVinci-Env 将 12.8K+ 仓库转为 45,320 个环境，并从约 9K 环境采集 13K 轨迹。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2603.13023","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/OpenSWE"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GAIR/OpenSWE"}],"link_count":3,"sections":9},{"id":"debuglm-2026","title":"DebugLM: Learning Traceable Training Data Provenance for LLMs","year":2026,"venue":"arXiv preprint","authors":["Wenjie Jacky Mo","Qin Liu","Xiaofei Wen","Wenxuan Zhou","Zhe Zhao","Muhao Chen"],"authors_zh":"Wenjie Jacky Mo、Qin Liu、Xiaofei Wen、Wenxuan Zhou、Zhe Zhao、Muhao Chen","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","audit","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","release_audit"],"domains":["data-provenance","safety","factual-qa","medical-qa","hazardous-knowledge"],"tags":["data-provenance","lineage","source-tags","multi-stage-training","traceability","test-time-remediation","targeted-refusal","audit"],"status":"partial","priority":"可读","paper_type_zh":"训练内溯源与修复配方","best_for_zh":"关注训练数据谱系、安全修复、模型自报和审计边界的研究者","confidence":"high","one_line":["DebugLM augments ordinary post-training records with dataset-level provenance and quarantine labels so a trained model can self-report a source tag or selectively refuse source-associated behavior.","DebugLM 用数据集级溯源与隔离标签增强 SFT/GRPO 记录，使模型能自报来源或针对来源选择性拒答。"],"why":"It is a concrete construction pattern for carrying source identity through post-training into deployment-time diagnosis, while exposing the audit gap between a trained self-report and independently replayable lineage.","primary_link":"https://arxiv.org/abs/2603.17884","links":[],"link_count":2,"sections":9},{"id":"deepresearch-bench-ii-expert-reports-2026","title":"DeepResearch Bench II: Diagnosing Deep Research Agents via Rubrics from Expert Report","year":2026,"venue":"arXiv","authors":["Ruizhe Li","Mingxuan Du","Benfeng Xu","Chiwei Zhu","Xiaorui Wang","Zhendong Mao"],"authors_zh":"Ruizhe Li, Mingxuan Du, Benfeng Xu, Chiwei Zhu, Xiaorui Wang, Zhendong Mao","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["deep_research","web","cross_domain"],"tags":["deep_research","web","rubric","llm_judge"],"status":"verified","priority":"必读","paper_type_zh":"专家报告驱动的深度研究 Agent rubric 评测基准","best_for_zh":"研究深度研究 Agent、基于证据的 LLM Judge 与长文报告评测的读者。","confidence":"high","one_line":["DeepResearch Bench II scores deep-research reports with 9,430 expert-grounded binary rubrics across retrieval, analysis, and presentation.","DeepResearch Bench II 用源自专家报告的 9,430 条二元准则诊断深度研究 Agent 的检索、分析与呈现。"],"why":"It grounds long-form research evaluation in verifiable expert evidence rather than ad hoc model criteria.","primary_link":"https://arxiv.org/abs/2601.08536","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/imlrz/DeepResearch-Bench-II"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/muset-ai/DeepResearch-Bench-II-Dataset"}],"link_count":4,"sections":9},{"id":"deepresearch-9k-deep-research-agent-2026","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","year":2026,"venue":"SIGIR 2026","authors":["Tongzhou Wu","Yuhao Wang","Xinyu Ma","Xiuqiang He","Shuaiqiang Wang","Dawei Yin","Xiangyu Zhao"],"authors_zh":"Tongzhou Wu, Yuhao Wang, Xinyu Ma, Xiuqiang He, Shuaiqiang Wang, Dawei Yin, Xiangyu Zhao","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["web-research","retrieval-agents","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"深度研究智能体的分级检索轨迹、答案验证与训练框架论文","best_for_zh":"需要长程搜索、证据整合与难度可控过程轨迹来训练或评测深度研究智能体的研究者。","confidence":"high","one_line":["DeepResearch-9K releases 9,000 difficulty-tiered web-research questions with teacher search traces, reasoning chains, and verifiable answers.","DeepResearch-9K 提供按搜索深度分级的 9,000 个深度研究问题、可验证答案及带推理链的多轮检索轨迹。"],"why":"It couples long-horizon retrieval traces with an explicit, controllable difficulty signal tied to search effort.","primary_link":"https://arxiv.org/pdf/2603.01152","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/artillerywu/DeepResearch-9K"}],"link_count":5,"sections":9},{"id":"deepvision-103k-a-large-scale-challenging-decontaminated-and-verifiable-multimodal-mathe","title":"DeepVision-103K: A Visually Diverse, Broad-Coverage, and Verifiable Mathematical Dataset for Multimodal Reasoning","year":2026,"venue":"arXiv","authors":["Haoxiang Sun","Lizhen Xu","Bing Zhao","Wotao Yin","Wei Wang","Boyu Yang","Rui Wang","Hu Wei"],"authors_zh":"Haoxiang Sun, Lizhen Xu, Bing Zhao, Wotao Yin, Wei Wang, Boyu Yang, Rui Wang, Hu Wei","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** DeepVision-103K provides about 103K visually diverse, decontaminated K–12 math problems with rule-verifiable final answers.","DeepVision-103K 提供约 10.3 万道视觉多样、去污染且最终答案可规则验证的 K–12 数学题。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2602.16742","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SKYLENAGE-AI/DeepVision-103K"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/skylenage/DeepVision-103K"}],"link_count":3,"sections":9},{"id":"defab-defeasible-abduction-benchmark-2026","title":"DeFAb: A Verifiable Benchmark for Defeasible Abduction in Foundation Models","year":2026,"venue":"arXiv","authors":["Patrick Cooper","Alvaro Velasquez"],"authors_zh":"Patrick Cooper, Alvaro Velasquez","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","rlvr","preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["logical-reasoning","formal-mathematics","knowledge-reasoning"],"tags":["defeasible-abduction","logical-reasoning","verifier","rlvr","2026"],"status":"verified","priority":"可读","paper_type_zh":"可程序验证的逻辑推理基准与数据构建方法","best_for_zh":"需要精确逻辑奖励、溯因推理评测或可验证推理训练数据的研究者。","confidence":"high","one_line":["DeFAb turns public knowledge bases into more than 372,000 defeasible-abduction tasks whose hypotheses are checked for derivation, conservativity, and minimality.","DeFAb 将公开知识库转化为 37 万余个可判定溯因任务，并以规则求解器检查假设的推导性、保守性与最小性。"],"why":"It makes minimal theory revision an executable acceptance condition instead of relying on a subjective language-model judge.","primary_link":"https://arxiv.org/abs/2606.18557","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PatrickAllenCooper/DeFAb"}],"link_count":3,"sections":9},{"id":"demystifying-multi-agent-debate-2026","title":"Demystifying Multi-Agent Debate: The Role of Confidence and Diversity","year":2026,"venue":"Findings of ACL 2026","authors":["Xiaochen Zhu","Caiqi Zhang","Yizhou Chi","Tom Stafford","Nigel Collier","Andreas Vlachos"],"authors_zh":"Xiaochen Zhu、Caiqi Zhang、Yizhou Chi、Tom Stafford、Nigel Collier、Andreas Vlachos","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["reasoning","question_answering","multi_agent"],"tags":["multi-agent-debate","confidence","diversity","grpo"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["DMAD samples 10 candidates, selects five distinct answers, and trains five-turn homogeneous debate to express and use 0-10 confidence with correctness, calibration, engagement, and format rewards, while releasing code but not full traces.","DMAD 采样 10 个候选、挑出五个互异答案，训练五轮同质辩论来表达并使用 0 到 10 的置信度，奖励涵盖正确性、校准度、参与度与格式；开源了代码但未发布完整轨迹。"],"why":"It makes candidate coverage, confidence communication, reward components, and belief updates explicit trace fields for auditing multi-agent test-time search.","primary_link":"https://aclanthology.org/2026.findings-acl.1694/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SpaceHunterInf/DMAD"}],"link_count":5,"sections":9},{"id":"codec-contamination-context-2026","title":"Detecting Data Contamination in LLMs via In-Context Learning","year":2026,"venue":"ICLR 2026 Poster","authors":["Michał Zawalski","Meriem Boubdir","Klaudia Bałazy","Besmira Nushi","Pablo Ribalta"],"authors_zh":"Michał Zawalski, Meriem Boubdir, Klaudia Bałazy, Besmira Nushi, Pablo Ribalta","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"数据污染检测与基准审计方法","best_for_zh":"需要在已知或未知预训练语料下审计语言模型评测污染的研究者。","confidence":"high","one_line":["CoDeC measures the change induced by in-context examples to identify memorized benchmark data.","以同分布上下文是否降低 token 置信度，量化语言模型对候选数据集的污染风险。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://openreview.net/forum?id=YlpaaYxx4t","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=YlpaaYxx4t&name=supplementary_material"}],"link_count":3,"sections":9},{"id":"dvd-variant-contamination-2026","title":"Detecting Variant Contamination in LLMs via Variance of Generation Distribution","year":2026,"venue":"ICLR 2026 withdrawn","authors":["Renzhao Liang","Jingru Chen","Bo Jia","Yidong Wang","Jin Ke","Linfeng Zhang","Xin Wang","Bo Deng","Cunxiang Wang"],"authors_zh":"Renzhao Liang, Jingru Chen, Bo Jia, Yidong Wang, Jin Ke, Linfeng Zhang, Xin Wang, Bo Deng, Cunxiang Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"变体数据污染检测方法","best_for_zh":"需要审计非逐字基准泄漏的 LLM 评测研究者。","confidence":"high","one_line":["Variance-based detector targets paraphrased and structurally transformed leakage.","以生成分布方差检测释义与结构变体污染。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://openreview.net/forum?id=Ubi631nNbI","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=Ubi631nNbI&name=supplementary_material"}],"link_count":2,"sections":9},{"id":"difficulty-dpo-preference-selection-2026","title":"Difficulty-Based Preference Data Selection by DPO Implicit Reward Gap","year":2026,"venue":"arXiv preprint","authors":["Xuan Qi","Rongwu Xu","Zhijing Jin"],"authors_zh":"Xuan Qi、Rongwu Xu、Zhijing Jin（华盛顿大学、马克斯·普朗克智能系统研究所、多伦多大学等）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","reward-modeling"],"tags":["preference-data-selection","dpo","implicit-reward","difficulty"],"status":"verified","priority":"必读","paper_type_zh":"基于难度的偏好数据选择研究","best_for_zh":"适合在 DPO 或奖励模型训练前为现有偏好对排序、降低对齐训练成本的读者。","confidence":"high","one_line":["The method trains on preference pairs with the smallest DPO implicit reward gaps because those boundary cases provide the strongest learning signal.","该方法选择 DPO 隐式奖励差最小的偏好对，因为这些决策边界样本能提供最强的学习信号。"],"why":"It provides a simple, model-aware criterion for turning DPO's implicit reward into a reusable data-selection decision.","primary_link":"https://arxiv.org/abs/2508.04149","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Difficulty-Based-Preference-Data-Select"}],"link_count":3,"sections":9},{"id":"discoverllm-from-executing-intents-to-discovering-them-2026","title":"DiscoverLLM: From Executing Intents to Discovering Them","year":2026,"venue":"ICML 2026","authors":["Tae Soo Kim","Yoonjoo Lee","Jaesang Yu","John Joon Young Chung","Juho Kim"],"authors_zh":"Tae Soo Kim、Yoonjoo Lee、Jaesang Yu、John Joon Young Chung、Juho Kim","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"多轮对话偏好数据与意图发现对齐论文","best_for_zh":"研究对话式偏好学习、用户意图建模或交互式对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","DiscoverLLM 将潜在意图逐步具体化作为奖励信号，提供多轮偏好记录用于对话式偏好优化。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2602.03429","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tsook/discoverllm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/kixlab/DiscoverLLM-multiturn-preferences"},{"key":"project","label":["Project","项目主页"],"url":"https://taesookim.com/discoverllm/"}],"link_count":4,"sections":9},{"id":"feedback-memory-as-a-tool-2026","title":"Distilling Feedback into Memory-as-a-Tool","year":2026,"venue":"ICLR 2026 MemAgents Workshop","authors":["Víctor Gallego"],"authors_zh":"Víctor Gallego","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","sft","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["llm_judge","rubric_evaluation"],"tags":["rubric","llm_judge","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"反馈数据与 Agent 记忆论文","best_for_zh":"希望复用连续评审反馈并控制 Agent 推理成本的研究者。","confidence":"high","one_line":["Memory-as-a-Tool distils recurring rubric feedback into retrievable, updateable files to amortize refinement cost.","将重复 rubric 批评蒸馏为 Agent 可检索、更新的文件记忆，以降低后续改写成本。"],"why":"It exposes rubric-grounded evaluation signals and their audit boundary.","primary_link":"https://openreview.net/forum?id=hvfhz64q0O","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/vicgalle/feedback-memory-as-a-tool"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/vicgalle/rubric-feedback-bench"}],"link_count":5,"sections":9},{"id":"superior-reasoning-sft-2026","title":"Distribution-Aligned Sequence Distillation for Superior Long-CoT Reasoning","year":2026,"venue":"arXiv preprint","authors":["Yan, Shaotian","Liu, Kaiyuan","Shen, Chen","Wang, Bing","Fan, Sinan","Zhang, Jun","Wu, Yue","Wang, Zheng","Ye, Jieping"],"authors_zh":"Yan, Shaotian、Liu, Kaiyuan、Shen, Chen、Wang, Bing、Fan, Sinan、Zhang, Jun、Wu, Yue、Wang, Zheng、Ye, Jieping","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["distribution-aligned-compact-long-CoT-distillation"],"tags":["instruction-demonstration-rationale","arxiv-2601.09088","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"面向 4B 模型的长推理监督蒸馏","confidence":"high","one_line":["Distribution-aligned sequence distillation selects and stages 435K teacher traces to reduce this mismatch before training a 4B reasoning student.","该工作按序列分布对齐和温度阶段筛选约 43.5 万条长推理轨迹，用于训练 4B 学生。"],"why":"Small reasoning models often imitate long traces token by token even when the teacher and student sequence distributions are badly mismatched.","primary_link":"https://arxiv.org/abs/2601.09088","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Alibaba-Apsara/Superior-Reasoning-SFT-gpt-oss-120b"}],"link_count":2,"sections":9},{"id":"seal-vla-2025","title":"Do What You Say: Steering Vision-Language-Action Models via Runtime Reasoning-Action Alignment Verification","year":2026,"venue":"ICRA 2026","authors":["Yilin Wu","Anqi Li","Tucker Hermans","Fabio Ramos","Andrea Bajcsy","Claudia Pérez-D'Arpino"],"authors_zh":"Yilin Wu、Anqi Li、Tucker Hermans、Fabio Ramos、Andrea Bajcsy、Claudia Pérez-D'Arpino","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","verifier_reward","agent_environment"],"verification_contract":["environmental","judgment_required"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["embodied_ai","visual_reasoning","household_robotics"],"tags":["seal-vla","vla","runtime-steering","action-verification","libero"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["SEAL simulates candidate actions and selects the sequence best aligned with the VLA's textual plan.","SEAL 先模拟候选动作，再选出与视觉语言动作模型文字计划最一致的动作序列。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"It separately audits reasoning text, predicted physical outcome, and selected actions.","primary_link":"https://arxiv.org/abs/2510.16281","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVlabs/actalign"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/libero-r-datasets"},{"key":"project","label":["Project","项目主页"],"url":"https://yilin-wu98.github.io/steering-reasoning-vla/"}],"link_count":5,"sections":9},{"id":"dockerless-2026","title":"Dockerless: Environment-Free Program Verifier for Coding Agents","year":2026,"venue":"arXiv preprint","authors":["Wenhao Zeng","Yuling Shi","Xiaodong Gu","Chao Hu","Chaofan Wang","Yuhao Cui","Hongting Zhou","Mengnan Qi","Jianqiao Wangni","Zhaojian Yu","Shuzheng Gao","Kai Cai","Shilin He"],"authors_zh":"Wenhao Zeng、Yuling Shi、Xiaodong Gu、Chao Hu、Chaofan Wang、Yuhao Cui、Hongting Zhou、Mengnan Qi、Jianqiao Wangni、Zhaojian Yu、Shuzheng Gao、Kai Cai、Shilin He","tracks":["environment_agent_trajectory_data"],"source_role":["verifier_reward","construction_recipe"],"verification_contract":["judgment_required","programmatic","mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","reward_modeling","rlvr","agent_training","evaluation"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["software_engineering","repository_level_code","code_agents","agent_trajectories","environment_interaction"],"tags":["environment-agent-trajectory-data","software-engineering-agent","repository-level-code","environment-free-rollouts","agentic-verifier","reward-modeling","sft-filtering","rlvr","golden-patch-dependency","verifier-gaming-risk"],"status":"partial","priority":"可读","paper_type_zh":"软件工程 Agent 验证器与轨迹构建方法","best_for_zh":"研究软件工程 Agent 轨迹、学习型 verifier、SFT 筛选、RLVR 奖励与环境复现审计的读者","confidence":"high","one_line":["Dockerless replaces per-repository test execution during SWE-agent post-training with a golden-patch-conditioned verifier that asks 2-4 questions, gathers read-only repository evidence, and scores final patches for SFT filtering and GRPO.","Dockerless 以执行测试标签训练依赖 golden patch 的 Qwen3.5-9B agentic verifier，再用仓库证据评分从 16K 条环境精简 rollout 中筛出 4K 条用于 SFT，并为 GRPO 提供 dense reward。"],"why":"It exposes a concrete environment-free trajectory-to-reward recipe and strong controlled results, while showing that learned patch rewards remain dependent on reference solutions, compiler/dependency signals, calibration, replay metadata, and safety controls that are not released.","primary_link":"https://arxiv.org/abs/2606.28436v1","links":[],"link_count":3,"sections":9},{"id":"docksmith-2026","title":"DockSmith: Scaling Reliable Coding Environments via an Agentic Docker Builder","year":2026,"venue":"ICML 2026","authors":["Jiaran Zhang","Lu Ma","Yanhao Li","Fanqi Wan","Di Qi","Xin Wu","Zhewei Huang","Liangyu Chen","Yingwei Ma","Qi Han","Xiangyu Zhang"],"authors_zh":"Jiaran Zhang、Lu Ma、Yanhao Li、Fanqi Wan、Di Qi、Xin Wu、Zhewei Huang、Liangyu Chen、Yingwei Ma、Qi Han、Xiangyu Zhang","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","construction_recipe","agent_environment","model_report"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","search_substrate","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["software_engineering","code_agents","docker_environment_construction","agent_trajectories"],"tags":["environment-agent-trajectory-data","docker-building","agent-trajectories","execution-grounded","supervised-fine-tuning","success-filtering"],"status":"partial","priority":"必读","paper_type_zh":"软件工程环境构造、智能体轨迹数据发布与 SFT 配方","best_for_zh":"研究 SWE agent 环境、执行反馈、轨迹筛选、可重放性与数据许可审计的读者","confidence":"medium","one_line":["DockSmith converts test-backed GitHub pull requests into success-filtered, execution-grounded Docker-building conversations for agent SFT and releases 39,719 per-agent chat fragments linked to 2,876 instances.","DockSmith 将通过 Docker 构建与测试验证的多智能体环境构造交互筛成 SFT 数据；公开发布的是关联 2,876 个实例的 39,719 个按智能体拆分的聊天片段，而非 39,719 条完整 episode。"],"why":"It makes environment construction itself a reusable supervision source and shows that success-filtered Docker traces can transfer to broader software-agent tasks, while exposing important selection, replay, licensing, and security audit gaps.","primary_link":"https://openreview.net/pdf?id=tRbgtWwmHB","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/8sj7df9k8m5x8/docker_building_training"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/8sj7df9k8m5x8/docksmith"}],"link_count":7,"sections":9},{"id":"dont-judge-code-cover-2026","title":"Don’t Judge Code by Its Cover: On the Robustness of LLM-as-a-Judge for Code Evaluation","year":2026,"venue":"Findings of EACL 2026","authors":["Jiwon Moon","Yerin Hwang","Dongryeol Lee","Taegwan Kang","Yongil Kim","Kyomin Jung"],"authors_zh":"Jiwon Moon, Yerin Hwang, Dongryeol Lee, Taegwan Kang, Yongil Kim, Kyomin Jung","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"代码 LLM 判别器鲁棒性审计","best_for_zh":"需要用 LLM 进行无参考代码正确性评测的研究者与工程团队。","confidence":"high","one_line":["Audits superficial-code and presentation biases in LLM code evaluators.","审计代码 LLM 判别器对语义等价表层改写的偏差。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2026.findings-eacl.70/","links":[],"link_count":1,"sections":9},{"id":"dose-multimodal-selection-2026","title":"DOSE: Data Selection for Multi-Modal LLMs via Off-the-Shelf Models","year":2026,"venue":"Findings of ACL 2026","authors":["Biao Wu","Yiwu Zhong","Meng Fang","Ling Chen"],"authors_zh":"Biao Wu, Yiwu Zhong, Meng Fang, Ling Chen","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["multimodal","visual_instruction_tuning","reasoning"],"tags":["data-selection","multimodal","visual-instruction-tuning","weighted-sampling","training-efficiency"],"status":"verified","priority":"必读","paper_type_zh":"多模态指令数据选择研究","best_for_zh":"适合希望在训练专用选择器代价过高或可能接触目标语料时，构建高效视觉指令微调子集的读者。","confidence":"high","one_line":["DOSE selects visual instruction data with off-the-shelf text-quality and image-text-alignment scores, then uses distribution-aware sampling to retain quality without collapsing coverage.","DOSE 使用现成模型衡量文本质量和图文对齐度，再以分布感知的抽样保留质量而不牺牲覆盖范围。"],"why":"It makes the selection distribution itself part of the training-data design, rather than treating high score alone as sufficient evidence of a useful example.","primary_link":"https://aclanthology.org/2026.findings-acl.45/","links":[],"link_count":2,"sections":9},{"id":"dr-post-training-data-regularization-2026","title":"Dr. Post-Training: A Data Regularization Perspective on LLM Post-Training","year":2026,"venue":"arXiv preprint","authors":["Pingbang Hu","Xueshen Liu","Z. Morley Mao","Jiaqi W. Ma"],"authors_zh":"Pingbang Hu、Xueshen Liu、Z. Morley Mao、Jiaqi W. Ma（伊利诺伊大学厄巴纳—香槟分校、密歇根大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["instruction-following","alignment","reasoning"],"tags":["post-training","data-regularization","sft","rlhf","rlvr"],"status":"verified","priority":"必读","paper_type_zh":"后训练数据正则化与优化目标研究","best_for_zh":"适合需要同时利用少量高保真目标数据与大量通用数据的后训练研究者。","confidence":"high","one_line":["Dr. Post-Training uses general data as a feasible-set regularizer for target post-training updates.","该方法将通用训练数据视为目标后训练更新的可行方向约束，而非一次性筛选的样本池。"],"why":"It makes data consumption part of the optimization geometry rather than a one-time selection decision.","primary_link":"https://arxiv.org/abs/2605.07063","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TRAIS-Lab/Dr.Post-Training"}],"link_count":3,"sections":9},{"id":"draco-deep-research-2026","title":"DRACO: a Cross-Domain Benchmark for Deep Research Accuracy, Completeness, and Objectivity","year":2026,"venue":"arXiv","authors":["Joey Zhong","Hao Zhang","Clare Southern","Jeremy Yang","Thomas Wang","Kate Jung","Shu Zhang","Denis Yarats","Johnny Ho","Jerry Ma"],"authors_zh":"Joey Zhong, Hao Zhang, Clare Southern, Jeremy Yang, Thomas Wang 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["deep_research","web","cross_domain"],"tags":["deep_research","rubric","web"],"status":"verified","priority":"必读","paper_type_zh":"真实深度研究请求的跨域 rubric 评测基准","best_for_zh":"研究深度研究 Agent、真实用户任务和报告质量 Judge 的读者。","confidence":"high","one_line":["DRACO scores real deep-research reports for accuracy, completeness, objectivity, and citation quality.","DRACO 用来自真实深度研究使用模式的任务及四维 rubric，审查报告的准确、完整、客观和引用质量。"],"why":"It grounds report evaluation in diverse real-world research requests rather than synthetic prompts.","primary_link":"https://arxiv.org/abs/2602.11685","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/perplexity-ai/draco"}],"link_count":3,"sections":9},{"id":"dual-consensus-escaping-from-spurious-majority-in-unsupervised-rlvr-via-two-stag-2026","title":"Dual Consensus: Escaping from Spurious Majority in Unsupervised RLVR via Two-Stage Vote Mechanism","year":2026,"venue":"arXiv preprint arXiv:2603.16223","authors":["Kaixuan Du","Meng Cao","Hang Zhang","Yukun Wang","Xiangzhou Huang","Ni Li"],"authors_zh":"Kaixuan Du、Meng Cao、Hang Zhang、Yukun Wang、Xiangzhou Huang、Ni Li","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🧯 Audit, Failure, Contamination, and Verifier Attacks map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","用双阶段共识缓解无监督 RLVR 中多数错误答案带来的伪信号，适合审计投票机制的失效条件。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://arxiv.org/abs/2603.16223","links":[],"link_count":1,"sections":9},{"id":"dynamic-cheatsheet-test-time-memory-2026","title":"Dynamic Cheatsheet: Test-Time Learning with Adaptive Memory","year":2026,"venue":"EACL 2026 Long Papers","authors":["Mirac Suzgun","Mert Yuksekgonul","Federico Bianchi","Dan Jurafsky","James Zou"],"authors_zh":"Mirac Suzgun、Mert Yuksekgonul、Dan Jurafsky（斯坦福大学）；Federico Bianchi（Together AI）；James Zou（斯坦福大学、Together AI）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","test-time-learning","memory","self-curation","black-box-llm"],"status":"verified","priority":"必读","paper_type_zh":"自适应记忆测试时学习研究","best_for_zh":"研究无需参数更新、却能让多次推理努力跨任务累积的读者。","confidence":"high","one_line":["Dynamic Cheatsheet uses self-curated persistent memory so a black-box model can reuse lessons discovered during earlier inference episodes.","Dynamic Cheatsheet 通过自整理的持久记忆，让黑箱模型复用先前推理过程中发现的策略、代码和纠错经验。"],"why":"It expands test-time scaling from repeated attempts on one prompt to accumulated, auditable experience over a task stream.","primary_link":"https://aclanthology.org/2026.eacl-long.333/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/suzgunmirac/dynamic-cheatsheet"}],"link_count":4,"sections":9},{"id":"sai-dpo-self-aware-sampling-2026","title":"Dynamic Sampling that Adapts: Self-Aware Iterative Data Persistent Optimization for Mathematical Reasoning","year":2026,"venue":"Findings of ACL 2026","authors":["Jun Rao","Xuebo Liu","Hexuan Deng","Zepeng Lin","Zixiong Yu","Jiansheng Wei","Xiaojun Meng","Min Zhang"],"authors_zh":"Jun Rao, Xuebo Liu, Hexuan Deng, Zepeng Lin, Zixiong Yu, Jiansheng Wei, Xiaojun Meng, Min Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning","rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics"],"tags":["dynamic-sampling","preference-optimization","mathematical-reasoning","dpo"],"status":"verified","priority":"必读","paper_type_zh":"用于离线偏好优化的动态数学推理数据选择","best_for_zh":"为迭代数学推理训练构建学习者自适应偏好数据集的读者。","confidence":"high","one_line":["SAI-DPO repeatedly selects preference-training records using model-relative knowledge weaknesses and self-aware difficulty.","SAI-DPO 按模型相对的知识弱点和自感知难度反复选择偏好训练记录。"],"why":"It turns changing learner competence into an explicit policy for which preference pairs enter the next optimization round.","primary_link":"https://aclanthology.org/2026.findings-acl.1412/","links":[],"link_count":2,"sections":9},{"id":"dynamicfocalpo-adaptive-preference-optimization-2026","title":"DynamicFocalPO: Adaptive Focusing Strategy for Preference Optimization","year":2026,"venue":"Findings of ACL 2026","authors":["Shu Zhou","Junan Chen","Rui Ling","Xin Wang","Tao Fan","Hao Wang"],"authors_zh":"Shu Zhou, Junan Chen, Rui Ling, Xin Wang, Tao Fan, Hao Wang（南京大学、百度、南京财经大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","curriculum-learning","sample-weighting","dpo","alignment"],"status":"verified","priority":"可读","paper_type_zh":"基于课程学习的偏好优化研究","best_for_zh":"适合拥有离线偏好数据集，并希望以轻量方式控制含糊或排序错误的回答对何时影响 DPO 训练的读者。","confidence":"high","one_line":["DynamicFocalPO schedules preference-pair weighting from easy, already-correctly-ranked records toward harder records as training progresses.","DynamicFocalPO 随训练推进调度偏好回答对的权重：先侧重已正确排序的容易记录，再逐步纳入困难记录。"],"why":"It makes training time an explicit part of how the optimizer consumes preference records instead of assigning every pair a fixed influence for the entire run.","primary_link":"https://aclanthology.org/2026.findings-acl.1009/","links":[],"link_count":2,"sections":9},{"id":"dynamics-constrained-mixtures-2026","title":"DynaMiCS: Fine-tuning LLMs with Performance Constraints using Dynamic Mixtures","year":2026,"venue":"arXiv preprint","authors":["Eleonora Gualdoni","Sonia Laguna","Louis Béthune","Joao Monteiro","Pierre Ablin","Marco Cuturi"],"authors_zh":"Eleonora Gualdoni, Sonia Laguna, Louis Béthune, Joao Monteiro, Pierre Ablin, Marco Cuturi","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","safety_alignment"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["instruction-tuning","reasoning","mathematics","code","safety"],"tags":["data-mixing","constraint-aware","sft","forgetting"],"status":"verified","priority":"必读","paper_type_zh":"带约束的动态监督微调数据混合优化研究","best_for_zh":"在专门化数据与明确安全或能力保持约束间取平衡的读者。","confidence":"high","one_line":["DynaMiCS repeatedly probes data-source effects and optimizes SFT mixture weights to improve targets without violating protected capabilities.","DynaMiCS 反复探测数据来源效应并优化监督微调混合权重，在改善目标的同时不违反受保护能力。"],"why":"It makes protected evaluation domains part of the data-consumption decision.","primary_link":"https://arxiv.org/abs/2605.10770","links":[],"link_count":2,"sections":9},{"id":"dynamixsft-instruction-mixtures-2026","title":"DYNAMIX SFT: Dynamic Mixture Optimization of Instruction Tuning Collections","year":2026,"venue":"Findings of ACL 2026","authors":["Haebin Shin","Lei Ji","Xiao Liu","Zhiwei Yu","Hyunwoo Yoo","Qi Chen","Yeyun Gong"],"authors_zh":"Haebin Shin, Lei Ji, Xiao Liu, Zhiwei Yu, Hyunwoo Yoo, Qi Chen, Yeyun Gong","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["instruction-tuning","reasoning","mathematics","code"],"tags":["data-mixing","instruction-tuning","sft","dynamic-sampling"],"status":"verified","priority":"必读","paper_type_zh":"动态指令数据混合优化研究","best_for_zh":"设计自适应来源级指令微调抽样的读者。","confidence":"high","one_line":["DYNAMIX SFT updates source-level instruction-data mixture weights from one-step learning progress while preserving coverage with a prior and exploration floor.","DYNAMIX SFT 以单步学习进度更新来源级指令数据混合权重，并用先验和探索下限保留覆盖面。"],"why":"It makes the training-time decision over data-source allocation explicit and measurable.","primary_link":"https://aclanthology.org/2026.findings-acl.1972/","links":[],"link_count":3,"sections":9},{"id":"easyrl-progressive-self-training-2026","title":"Easy Samples Are All You Need: Self-Evolving LLMs via Data-Efficient Reinforcement Learning","year":2026,"venue":"Findings of ACL 2026","authors":["Zhiyin Yu","Bo Zhang","Qibin Hou","Zhonghai Wu","Xiao Luo","Lei Bai"],"authors_zh":"Zhiyin Yu, Bo Zhang, Qibin Hou, Zhonghai Wu, Xiao Luo, Lei Bai","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics","science"],"tags":["rlvr","pseudo-labeling","curriculum","data-selection","self-training"],"status":"verified","priority":"必读","paper_type_zh":"渐进式伪标注选择与强化学习研究","best_for_zh":"适合把小规模已验证种子集和大量困难无标注池转化为分阶段推理训练课程的读者。","confidence":"high","one_line":["EasyRL grows a GRPO training set from a few easy labels by selecting consistent or reflection-resolved pseudo-labels from progressively harder problems.","EasyRL 从少量简单标注题出发，逐步选择一致或经反思解决的伪标注难题来扩展 GRPO 训练集。"],"why":"It makes pseudo-label acceptance and deferral an explicit training-data decision instead of treating every unlabeled example as equally ready for RL.","primary_link":"https://aclanthology.org/2026.findings-acl.773/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YuZhiyin/EasyRL"}],"link_count":3,"sections":9},{"id":"fover-2026","title":"Efficient PRM Training Data Synthesis via Formal Verification","year":2026,"venue":"Findings of ACL 2026","authors":["Ryo Kamoi","Yusen Zhang","Nan Zhang","Sarkar Snigdha Sarathi Das","Ranran Haoran Zhang","Wenpeng Yin","Rui Zhang"],"authors_zh":"Ryo Kamoi, Yusen Zhang, Nan Zhang, Sarkar Snigdha Sarathi Das, Ranran Haoran Zhang, Wenpeng Yin, Rui Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","verifier_reward","process_supervision","construction_recipe","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","process_reward"],"training_use":["reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["formal_logic","theorem_proving","mathematical_reasoning","natural_language_reasoning"],"tags":["process-reward-model","formal-verification","step-level-labels","z3","isabelle","formal-logic","theorem-proving","verifier-training","formal-to-informal-transfer","best-of-k"],"status":"partial","priority":"必读","paper_type_zh":"形式验证驱动的过程监督数据合成","best_for_zh":"研究可程序验证过程标签、PRM 数据配方与形式到非形式迁移的读者","confidence":"medium","one_line":["FOVER-40K pairs formal reasoning traces with Boolean step labels produced by Z3 and Isabelle, then trains Llama- and Qwen-based PRMs for process-level verification and Best-of-K selection.","FoVer 用 Z3 与 Isabelle/HOL 为形式逻辑和形式证明轨迹生成二元步骤标签，组成 FOVER-40K 并训练用于步骤判错与 Best-of-K 选择的 PRM。"],"why":"It demonstrates a low-LLM-call route to process supervision while making formalization fidelity, prover wrappers, step-dependency assumptions, dataset versioning and informal-domain transfer explicit audit boundaries.","primary_link":"https://aclanthology.org/2026.findings-acl.403/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/psunlpgroup/FoVer"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ryokamoi/FoVer-FormalLogic-FormalProof-Qwen-2.5-7B-LastStepBalanced-40k"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/ryokamoi/fover"},{"key":"project","label":["Project","项目主页"],"url":"https://fover-prm.github.io/"}],"link_count":8,"sections":9},{"id":"trace-temporal-reasoning-aggregation-2026","title":"Efficient Test-Time Scaling via Temporal Reasoning Aggregation","year":2026,"venue":"Findings of ACL 2026","authors":["Jiakun Li","Xingwei He","Kefan Li","Hongzheng Chai","Hongyue Yu","Yuan Yuan"],"authors_zh":"Jiakun Li、Xingwei He、Kefan Li、Hongzheng Chai、Hongyue Yu、Yuan Yuan（机构：北京航空航天大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","early-exit","temporal-aggregation","overthinking","reasoning-efficiency"],"status":"verified","priority":"必读","paper_type_zh":"无需训练的时间早停研究","best_for_zh":"部署长推理轨迹安全自适应停止的读者。","confidence":"high","one_line":["TRACE halts reasoning when answer consistency and confidence remain stable across recent reasoning steps.","TRACE 在答案一致性和置信度跨近期推理步骤保持稳定时停止推理。"],"why":"It treats convergence as a temporal property and directly reduces token waste from overthinking.","primary_link":"https://arxiv.org/abs/2604.17304","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/qianfantianyuzhouzhou/TRACE"}],"link_count":3,"sections":9},{"id":"medssr-2026","title":"Eliciting Medical Reasoning with Knowledge-enhanced Data Synthesis: A Semi-Supervised Reinforcement Learning Approach","year":2026,"venue":"Findings of ACL 2026","authors":["Haolin Li, Shuyang Jiang, Ruipeng Zhang, Jiangchao Yao, Ya Zhang, Yanfeng Wang"],"authors_zh":"Haolin Li, Shuyang Jiang, Ruipeng Zhang, Jiangchao Yao, Ya Zhang, Yanfeng Wang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["knowledge-controlled-medical-reasoning-synthesis"],"tags":["instruction-demonstration-rationale","arxiv-2604.11547","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"医学推理监督微调与半监督强化学习","confidence":"high","one_line":["MedSSR controls question synthesis with rare-disease knowledge, lets the policy produce pseudo-labels, and combines self-supervised with human-anchored reinforcement learning.","MedSSR 用罕见病知识控制问题合成，再由策略模型产生伪标签，公开 4.3 万条医学推理记录。"],"why":"Medical reasoning data is expensive to distill and underrepresents rare diseases, so larger generic trace sets do not reliably improve the long tail.","primary_link":"https://arxiv.org/abs/2604.11547","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tdlhl/MedSSR-Synthetic-43K"}],"link_count":2,"sections":9},{"id":"embodied-reasoner-2026","title":"Embodied-Reasoner: Synergizing Visual Search, Reasoning, and Action for Embodied Interactive Tasks","year":2026,"venue":"ACL 2026","authors":["Wenqi Zhang","Mengna Wang","Gangao Liu","Huixin Xu","Yiwei Jiang","Yongliang Shen","Guiyang Hou","Zhe Zheng","Hang Zhang","Xin Li","Jiajun Liu","Weiming Lu","Peng Li","Yueting Zhuang"],"authors_zh":"Wenqi Zhang、Mengna Wang、Gangao Liu、Huixin Xu、Yiwei Jiang、Yongliang Shen、Guiyang Hou、Zhe Zheng、Hang Zhang、Xin Li、Jiajun Liu、Weiming Lu、Peng Li、Yueting Zhuang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","process_supervision","verifier_reward","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["step_level","state_action_level","full_episode"],"training_use":["sft","distillation","process_supervision"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["embodied_ai","visual_reasoning","household_robotics"],"tags":["embodied-reasoner","observation-thought-action","ai2-thor","multimodal-trajectories","rejection-sampling","reflection-tuning","self-exploration","state-action-supervision","visual-reasoning","open-data"],"status":"partial","priority":"必读","paper_type_zh":"具身多模态过程数据构造与开放发布","best_for_zh":"研究具身推理、环境验证、轨迹回放和多模态数据审计的读者","confidence":"high","one_line":["Embodied-Reasoner releases 9,390 AI2-THOR Observation-Thought-Action episodes built from GPT-4o imitation traces, environment-filtered self-exploration, and failure-aware reflection tuning, with about 64K images and 8M thought tokens.","Embodied-Reasoner 发布 9,390 条 AI2-THOR Observation-Thought-Action 轨迹，串联 GPT-4o 模仿数据、环境过滤自探索与失败感知反思调优。"],"why":"It provides an auditable example of turning simulator state checks and failed prefixes into multimodal process supervision, while showing why candidate attrition, alternative valid paths, trace faithfulness, replay, and release loadability must be checked separately from benchmark gains.","primary_link":"https://aclanthology.org/2026.acl-long.1910/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zwq2018/embodied_reasoner"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zwq2018/embodied_reasoner"},{"key":"project","label":["Project","项目主页"],"url":"https://embodied-reasoner.github.io/"}],"link_count":7,"sections":9},{"id":"mlr-structured-multi-level-modeling-2026","title":"Enhancing Language Model Reasoning with Structured Multi-Level Modeling","year":2026,"venue":"ICLR 2026 Poster","authors":["Siheng Xiong","Ali Payani","Faramarz Fekri"],"authors_zh":"Siheng Xiong、Ali Payani、Faramarz Fekri","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["hierarchical-reasoning","process-supervision","preference-optimization"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要层级轨迹监督、自动步骤描述，或研究长 CoT 信用分配的研究者。","confidence":"high","one_line":["MLR separates long reasoning into planner-generated cognitive subgoals and executor-generated details, with Step-DPO turning outcome feedback into scalable stepwise preferences.","MLR 将长推理拆为规划器生成的认知子目标和执行器生成的细节，并以 Step-DPO 把结果反馈转化为可扩展的步骤偏好。"],"why":"It releases structured step descriptors rather than only raw CoT, providing a concrete supervision interface for planner–executor reasoning.","primary_link":"https://openreview.net/forum?id=PlkzZhqBCd","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sxiong/MLR_structured_trajectory"}],"link_count":2,"sections":9},{"id":"envfactory-2026","title":"EnvFactory: Scaling Tool-Use Agents via Executable Environments Synthesis and Robust RL","year":2026,"venue":"arXiv preprint","authors":["Minrui Xu","Zilin Wang","Mengyi Deng","Zhiwei Li","Zhicheng Yang","Xiao Zhu","Yinhong Liu","Boyu Zhu","Baiyu Huang","Chao Chen","Heyuan Deng","Fei Mi","Lifeng Shang","Xingshan Zeng","Zhijiang Guo"],"authors_zh":"Minrui Xu, Zilin Wang, Mengyi Deng, Zhiwei Li, Zhicheng Yang, Xiao Zhu, Yinhong Liu, Boyu Zhu, Baiyu Huang, Chao Chen, Heyuan Deng, Fei Mi, Lifeng Shang, Xingshan Zeng, Zhijiang Guo","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["tool_use","agent_trajectories","mcp_environments","stateful_environments"],"tags":["envfactory","tool-use","agentic-rl","mcp","executable-environments","agent-trajectories","trajectory-synthesis","stateful-environments","composite-reward","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"可执行环境、agent trajectory 数据与复合 reward 方法论文","best_for_zh":"研究工具使用 agent 的 SFT、RLVR、可执行环境合成、trajectory reward 与数据发布审计的读者","confidence":"high","one_line":["EnvFactory releases source-grounded stateful MCP sandboxes, step-expanded SFT records, and turn-level RL records whose feedback mixes reference-call coverage with exact final-state checks, but the public artifacts lack an immutable paper-run binding and rejected-trajectory ledger.","EnvFactory 发布 source-grounded 的有状态 MCP sandbox、逐 step 展开的 SFT record 与逐 turn 的 RL record，并以 reference-call coverage 和终态精确匹配组成反馈；但 paper-run 绑定、失败轨迹账本、许可证与数据规模说明仍未闭合。"],"why":"It makes executable state, tool dependencies, simulated-user interaction, reference calls, and final-state feedback first-class post-training data, while exposing why row-unit reconciliation, reward semantics, source rights, failure retention, and version pinning must be audited separately from benchmark gains.","primary_link":"https://arxiv.org/abs/2605.18703","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LARK-AI-Lab/EnvFactory"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/LARK-Lab/envfactory"},{"key":"project","label":["Project","项目主页"],"url":"https://lark-ai-lab.github.io/envfactory.github.io/"}],"link_count":5,"sections":9},{"id":"envscaler-2026","title":"EnvScaler: Scaling Tool-Interactive Environments for LLM Agent via Programmatic Synthesis","year":2026,"venue":"arXiv preprint","authors":["Xiaoshuai Song","Haofei Chang","Guanting Dong","Yutao Zhu","Zhicheng Dou","Ji-Rong Wen"],"authors_zh":"Xiaoshuai Song, Haofei Chang, Guanting Dong, Yutao Zhu, Zhicheng Dou, Ji-Rong Wen","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["tool_use","agent_trajectories","stateful_environments","multi_turn_interaction","synthetic_environments"],"tags":["envscaler","environment-agent-trajectory-data","tool-use","executable-environment","synthetic-environment","agent-sft","agent-rlvr","terminal-state-verifier","mixed-verification","failure-retention-gap","version-drift","code-execution-risk"],"status":"partial","priority":"可读","paper_type_zh":"合成工具交互环境、训练轨迹与终态 verifier 发布","best_for_zh":"研究 agent SFT、RLVR、可执行环境扩展、终态判分与数据发布审计的读者","confidence":"high","one_line":["EnvScaler releases 191 executable tool environments, 7,234 SFT/RL scenarios, and 9,022 SFT trajectories, but SFT success is not terminal-state verified and reproducible RL reuse still lacks rollout, version, and verifier-audit records.","EnvScaler 发布 191 个可执行工具环境、7,234 条 SFT/RL scenario 和 9,022 条 SFT trajectory，并以终态 Python check 支持 RLVR；但 RL rollout 未发布、SFT 不做终态正确性验证，默认 HF loader、代码隔离和版本回放仍有缺口。"],"why":"It makes environment scale a concrete post-training data variable—state, tools, rules, tasks, trajectories, and executable outcome checks—while showing why release shape, terminal-versus-success semantics, rejected samples, code safety, and version pins determine whether synthetic agent data can be trusted or replayed.","primary_link":"https://arxiv.org/abs/2601.05808","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUC-NLPIR/EnvScaler"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/XXHStudyHard/envscaler"}],"link_count":12,"sections":9},{"id":"euclean-automated-geometry-problem-formalization-with-unified-verification-in-lean","title":"Euclean: Automated Geometry Problem Formalization with Unified Verification in Lean","year":2026,"venue":"ICML 2026","authors":["Linbin Tang","Jingyan You","Zilin Kang","Hanzhang Liu","Sophia Zhang","Zenan Li","Chenrui Cao","Liangcheng Song","Jiaao Wu","Xian Zhang","Fan Yang"],"authors_zh":"Linbin Tang, Jingyan You, Zilin Kang, Hanzhang Liu, Sophia Zhang, Zenan Li, Chenrui Cao, Liangcheng Song, Jiaao Wu, Xian Zhang, Fan Yang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Hundreds of thousands of natural-language geometry problems are formalized in native Mathlib and verified uniformly by Lean.","把十万级自然语言几何题形式化到原生 Mathlib，并用 Lean 统一验证。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://openreview.net/forum?id=OUtgFOscnh","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tlb-22/Euclean"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tlb-22/euclean-numina-geometry"}],"link_count":3,"sections":9},{"id":"evaluating-the-robustness-of-proof-autoformalization-in-lean-4","title":"Evaluating the Robustness of Proof Autoformalization in Lean 4","year":2026,"venue":"arXiv","authors":["Zhengtao Gui","Sheng Yang","Zhouxing Shi"],"authors_zh":"Zhengtao Gui, Sheng Yang, Zhouxing Shi","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Global and local perturbations on miniF2F and MATH-500 test Lean autoformalization robustness.","在 miniF2F 与 MATH-500 上用全局和局部扰动测试 Lean 自动形式化鲁棒性。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2606.14867","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ucr-rai/robust-proof-autoformalization"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ucr-rai/RobustPABench"}],"link_count":3,"sections":9},{"id":"evm-questbench-transaction-code-generation-2026","title":"EVM-QuestBench: An Execution-Grounded Benchmark for Natural-Language Transaction Code Generation","year":2026,"venue":"ACL 2026","authors":["Pei Yang","Wanyi Chen","Ke Wang","Lynn Ai","Eric Yang","Tianyu Shi"],"authors_zh":"Pei Yang, Wanyi Chen, Ke Wang, Lynn Ai, Eric Yang, Tianyu Shi","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["evaluation","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code","reasoning"],"tags":["evm","blockchain","transaction-code","execution","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"带链状态验证器的自然语言交易代码生成基准","best_for_zh":"研究区块链交易代理、执行式评测或基于状态奖励训练的研究者。","confidence":"high","one_line":["EVM-QuestBench scores generated transaction scripts by the on-chain states they achieve on forked EVM snapshots.","EVM-QuestBench 在分叉 EVM 链上以终局状态验证生成的自然语言交易脚本。"],"why":"It replaces string-level transaction-code checks with programmatic validation of asset and contract state.","primary_link":"https://aclanthology.org/2026.acl-long.1642/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenEdgeHQ/EVM-quest-bench"}],"link_count":4,"sections":9},{"id":"exp-bench-ai-research-experiments-2026","title":"EXP-Bench: Can AI Conduct AI Research Experiments?","year":2026,"venue":"ICLR 2026","authors":["Patrick Tser Jern Kon","Qiuyi Ding","Jiachen Liu","Xinyi Zhu","Jingjia Peng","Jiarong Xing","Yibo Huang","Yiming Qiu","Jayanth Srinivasa","Myungjin Lee","Mosharaf Chowdhury","Matei Zaharia","Ang Chen"],"authors_zh":"Patrick Tser Jern Kon, Qiuyi Ding, Jiachen Liu, Xinyi Zhu, Jingjia Peng, Jiarong Xing, Yibo Huang, Yiming Qiu, Jayanth Srinivasa, Myungjin Lee, Mosharaf Chowdhury, Matei Zaharia, Ang Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","step_level"],"training_use":["evaluation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["ai-research","machine-learning","agent-evaluation"],"tags":["ai-research","experiments","agents","execution-verification","2026"],"status":"verified","priority":"可读","paper_type_zh":"端到端 AI 研究实验的可执行智能体基准","best_for_zh":"需要评测智能体能否完成完整机器学习实验流程的研究者。","confidence":"high","one_line":["EXP-Bench turns 51 AI papers into 461 executable research-experiment tasks requiring agents to design, implement, run, and analyze experiments.","EXP-Bench 从 51 篇 AI 论文构建 461 个可执行研究实验任务，要求智能体完成设计、实现、运行与结果分析。"],"why":"It evaluates research agents through runnable experimental artifacts rather than a prose account of a plan.","primary_link":"https://arxiv.org/abs/2505.24785","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Just-Curieous/Curie/tree/main/benchmark/exp_bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Just-Curieous/EXP-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://amberljc.github.io/blog/2025-05-11-exp-bench.html"}],"link_count":6,"sections":9},{"id":"reagent-reasoning-reward-agents-2026","title":"Exploring Reasoning Reward Model for Agents","year":2026,"venue":"ACL 2026 Findings","authors":["Baixuan Fan","Maituo Feng","Panyuan Zhang","Tianshuo Peng","Shixun Li","Filei Jiang","Shawn Chen","Feng Pei","Sunliang Cai","Xiangyu Yue"],"authors_zh":"Baixuan Fan、Maituo Feng、Panyuan Zhang、Tianshuo Peng、Shixun Li、Filei Jiang、Shawn Chen、Feng Pei、Sunliang Cai、Xiangyu Yue","tracks":["preference_reward_feedback_data","process_trace_supervision_data","training_usage_optimization_objectives"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"智能体推理奖励模型与偏好数据论文","best_for_zh":"研究偏好数据、奖励建模或对齐的读者。","confidence":"medium","one_line":["The paper provides preference or reward feedback data for alignment research.","ReAgent 探索用于智能体的推理奖励模型，并发布相应的奖励训练数据。"],"why":"It makes a feedback object available for alignment training, evaluation, or analysis.","primary_link":"https://arxiv.org/abs/2601.22154","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kxfan2002/Reagent"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/bunny127/Reagent-RRM-RL-90K"}],"link_count":5,"sections":9},{"id":"faithcot-bench-instance-level-faithfulness-2025","title":"FaithCoT-Bench: Benchmarking Instance-Level Faithfulness of Chain-of-Thought Reasoning","year":2026,"venue":"ICLR 2026","authors":["Xu Shen","Song Wang","Zhen Tan","Laura Yao","Xinyu Zhao","Kaidi Xu","Xin Wang","Tianlong Chen"],"authors_zh":"Xu Shen、Song Wang、Zhen Tan、Laura Yao、Xinyu Zhao、Kaidi Xu、Xin Wang、Tianlong Chen","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["chain-of-thought-faithfulness","reasoning-evaluation","process-supervision"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测 CoT 忠实性、比较过程检测器或分析后验解释与真实决策脱节的研究者。","confidence":"high","one_line":["FaithCoT-Bench pairs expert-annotated CoT trajectories with step evidence to test whether a specific explanation reflects a model's internal reasoning.","FaithCoT-Bench 以带步骤证据的专家标注 CoT 轨迹，检验某条解释是否反映模型的内部推理。"],"why":"It operationalizes instance-level CoT faithfulness with expert evidence instead of inferring trustworthiness from final-answer correctness.","primary_link":"https://arxiv.org/abs/2510.04040","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/se7esx/FaithCoT-BENCH"},{"key":"data","label":["Data","数据"],"url":"https://github.com/se7esx/FaithCoT-BENCH/blob/main/faithcot.zip"}],"link_count":5,"sections":9},{"id":"bonafide-faithfulness-meta-evaluation-2026","title":"Faithfulness Metrics Don't Measure Faithfulness: A Meta-Evaluation with Ground Truth","year":2026,"venue":"arXiv preprint","authors":["Yoav Gur-Arieh","Ana Marasović","Mor Geva"],"authors_zh":"Yoav Gur-Arieh、Ana Marasović、Mor Geva","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["chain-of-thought-faithfulness","meta-evaluation","process-supervision"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要验证 CoT 忠实性指标、构造可控过程标签或审计解释度量是否真正有效的研究者。","confidence":"high","one_line":["BonaFide creates CoTs with recoverable ground-truth computation labels to show that common faithfulness metrics often measure neither step nor whole-trace faithfulness reliably.","BonaFide 构造可恢复真实计算标签的 CoT，用以证明常见忠实性指标往往无法可靠衡量步骤或整条轨迹的忠实性。"],"why":"It replaces plausibility and importance proxies with task designs that expose a known required computation for direct metric validation.","primary_link":"https://arxiv.org/abs/2605.25052","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yoavgurarieh/BonaFide"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yoavgurarieh/BonaFide-Extended"}],"link_count":4,"sections":9},{"id":"fasttts-edge-test-time-scaling-2026","title":"FastTTS: Accelerating Test-Time Scaling for Edge LLM Reasoning","year":2026,"venue":"ASPLOS 2026","authors":["Hao Mark Chen","Zhiwen Mo","Guanxi Lu","Shuang Liang","Lingxiao Ma","Wayne Luk","Hongxiang Fan"],"authors_zh":"Hao Mark Chen、Zhiwen Mo、Guanxi Lu、Shuang Liang、Lingxiao Ma、Wayne Luk、Hongxiang Fan（机构：帝国理工学院、微软研究院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","edge-serving","verifier-guided-search","speculative-execution","kv-cache"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展推理服务系统论文","best_for_zh":"在严格 GPU 内存与延迟约束下部署验证器引导推理搜索的读者。","confidence":"high","one_line":["FastTTS accelerates verifier-guided test-time reasoning on a single edge GPU through speculative beam extension, prefix-aware scheduling, and asymmetric generator-verifier memory allocation.","FastTTS 通过推测式束扩展、前缀感知调度和生成器—验证器的非对称内存分配，在单张边缘 GPU 上加速验证器引导的测试时推理。"],"why":"It treats test-time scaling as a systems scheduling problem and shows that the same search policy can be substantially faster without changing its algorithmic output.","primary_link":"https://arxiv.org/abs/2509.00195","links":[],"link_count":2,"sections":9},{"id":"feedback-driven-tool-use-improvements-via-automated-build-environments-2025","title":"Feedback-Driven Tool-Use Improvements in Large Language Models via Automated Build Environments","year":2026,"venue":"Findings of the Association for Computational Linguistics: ACL 2026","authors":["Junjie Ye","Changhao Jiang","Zhengyin Du","Yufei Xu","Xuesong Yao","Zhiheng Xi","Xiaoran Fan","Qi Zhang","Tao Gui","Xuanjing Huang","Jiecao Chen"],"authors_zh":"Junjie Ye, Changhao Jiang, Zhengyin Du, Yufei Xu, Xuesong Yao, Zhiheng Xi, Xiaoran Fan, Qi Zhang, Tao Gui, Xuanjing Huang, Jiecao Chen","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","agent_environment","construction_recipe","verifier_reward"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","scalar_reward"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","environment_interaction","tool_use","multi_step_question_answering"],"tags":["environment-agent-trajectory-data","tool-use","local-executable-environments","agentic-rl","rlvr","grpo","reinforce-plus-plus","environment-feedback","trajectory-release-incomplete","verifier-security-risk"],"status":"partial","priority":"可读","paper_type_zh":"可执行工具环境与反馈驱动智能体训练论文","best_for_zh":"研究环境型 agent trajectory、工具反馈、RLVR reward、可执行数据安全与发布审计的读者","confidence":"medium","one_line":["FTRL turns 2,415 synthetic local tool environments into current-policy interaction states scored by executable feedback for GRPO or Reinforce++, while leaving sampled trajectories, data rights, and secure paper-exact replay unresolved.","FTRL 将 2,415 个合成的本地 Python 工具环境转化为带可执行反馈和标量奖励的当前策略交互状态，用于 GRPO 或 Reinforce++；但论文运行轨迹、数据权利与安全可重放包均未公开。"],"why":"It exposes a concrete bridge from environment state and tool observations to scalar agent-training reward, and it also shows why an official code/data release must be audited for trajectory completeness, verifier brittleness, executable-code safety, licensing, and version replay before reuse.","primary_link":"https://aclanthology.org/2026.findings-acl.109/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bytedance/FTRL"},{"key":"data","label":["Data","数据"],"url":"https://github.com/bytedance/FTRL/tree/7022ef0529bbff4901084f17e452eb49cad5fffc/Data/jsonl/raw"}],"link_count":5,"sections":9},{"id":"fot-fine-grained-data-ordering-2026","title":"Fine-Grained Data Ordering Improves Fine-Tuning for Large Language Models","year":2026,"venue":"Findings of ACL 2026","authors":["Xiaomeng Hu","Yixuan Tang","Haoze Li","Hao Chen","Qi Zhang","Zhanming Shen","Yiming Zhang","Haobo Wang","Junbo Zhao"],"authors_zh":"Xiaomeng Hu, Yixuan Tang, Haoze Li, Hao Chen, Qi Zhang, Zhanming Shen, Yiming Zhang, Haobo Wang, Junbo Zhao","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["continued-pretraining","instruction-tuning"],"tags":["curriculum-learning","data-ordering","fine-tuning","continued-pretraining","difficulty"],"status":"verified","priority":"必读","paper_type_zh":"细粒度训练数据调度研究","best_for_zh":"为持续预训练或监督微调设计课程顺序、但不希望丢弃样本的读者。","confidence":"high","one_line":["FOT scores examples by model-relative difficulty, trains easy-to-hard in the first epoch, and restores random order thereafter.","FOT 按模型相对难度为样本打分，在第一轮按由易到难训练，之后恢复随机顺序。"],"why":"It shows that the order in which data is consumed is itself a training decision, distinct from filtering or mixing the data beforehand.","primary_link":"https://aclanthology.org/2026.findings-acl.1021/","links":[],"link_count":2,"sections":9},{"id":"fineverify-agentic-search-2026","title":"FineVerify: Scaling Test-Time Compute with Fine-Grained Self-Verification for Agentic Search","year":2026,"venue":"arXiv preprint","authors":["James Xu Zhao","Hui Chen","Bryan Hooi","See-Kiong Ng"],"authors_zh":"James Xu Zhao、Hui Chen、Bryan Hooi、See-Kiong Ng（新加坡国立大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","step_level"],"training_use":["test_time_compute","evaluation","audit"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-search","information-seeking"],"tags":["test-time-compute","agentic-search","self-verification","reranking","web-agents"],"status":"verified","priority":"可读","paper_type_zh":"面向智能体搜索的细粒度测试时验证研究","best_for_zh":"设计多证据问答、网页搜索智能体与轨迹选择器的读者。","confidence":"high","one_line":["FineVerify turns each sampled agent trajectory into sub-question-level checks, enabling a more reliable selection policy as the test-time trajectory budget grows.","FineVerify 将每条搜索轨迹按可检查的子问题逐项验证，再用聚合得分选择答案，从而提升轨迹扩展的有效性。"],"why":"It shows that adding trajectories is useful only when the selector can inspect their multi-claim evidence rather than rely on a single confidence score.","primary_link":"https://arxiv.org/abs/2606.00660","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XuZhao0/fineverify"}],"link_count":3,"sections":9},{"id":"fingertip-20k-2025","title":"FingerTip 20K: A Benchmark for Proactive and Personalized Mobile LLM Agents","year":2026,"venue":"ICLR 2026 Poster","authors":["Qinglong Yang","Haoming Li","Haotian Zhao","Xiaokai Yan","Jingtao Ding","Fengli Xu","Yong Li"],"authors_zh":"Qinglong Yang, Haoming Li, Haotian Zhao, Xiaokai Yan, Jingtao Ding, Fengli Xu, Yong Li","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["agent_trajectories","mobile_gui","proactive_task_suggestion","personalized_task_execution"],"tags":["environment-agent-trajectory-data","mobile-gui-agent","human-demonstrations","longitudinal-user-context","proactive-task-suggestion","personalized-execution","mixed-verification","privacy-risk"],"status":"partial","priority":"必读","paper_type_zh":"移动 GUI 智能体纵向轨迹数据集与混合反馈基准","best_for_zh":"研究移动智能体 SFT、个性化交互评测、轨迹发布审计、隐私风险与 live environment 复现边界的读者","confidence":"medium","one_line":["A longitudinal Android benchmark of human state-action episodes for intent anticipation and preference-conditioned execution, with a verified gap between the paper's 21,437-episode corpus and the public 20,000-row index.","FingerTip 20K 以纵向用户画像和真实 Android 状态—动作示范支持主动意图预测与个性化执行，但论文的 21,437 个 episode 与公开索引的 20,000 行无法对齐，终态成功还依赖未编码进脚本的人工判断。"],"why":"It shows how user profiles, time, location, intent history, screenshots, accessibility trees, and human action traces can become both supervised agent data and a benchmark surface; it also exposes why episode counts, manual terminal judgments, privacy filtering, and live-app replay metadata must be audited before reuse.","primary_link":"https://openreview.net/pdf?id=n3iFV0gLMc","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tsinghua-fib-lab/FingerTip-20K"},{"key":"data","label":["Data","数据"],"url":"https://www.kaggle.com/datasets/qinglongyang/fingertip-20k/data"}],"link_count":7,"sections":9},{"id":"fol-traces-verified-first-order-logic-reasoning-traces-2026","title":"FOL-Traces: Verified First-Order Logic Reasoning Traces at Scale","year":2026,"venue":"Findings of the Association for Computational Linguistics: EACL 2026","authors":["Isabelle Lee","Sarah Liaw","Dani Yogatama"],"authors_zh":"Isabelle Lee、Sarah Liaw、Dani Yogatama","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["formal_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"程序化验证的一阶逻辑推理轨迹数据与诊断基准论文","best_for_zh":"研究可验证逻辑推理过程、步骤补全和掩码操作预测。","confidence":"high","one_line":["FOL-Traces releases programmatically verified first-order-logic reasoning traces with complexity metadata for diagnosing structured inference.","FOL-Traces 发布带复杂度元数据的程序化验证一阶逻辑推理轨迹，用于诊断结构化推断能力。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://aclanthology.org/2026.findings-eacl.115/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/iglee/fol-traces"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/fol-traces/fol-traces"}],"link_count":5,"sections":9},{"id":"w2bench-wrl-2026","title":"From Coarse to Fine: Benchmarking and Reward Modeling for Writing-Centric Generation Tasks","year":2026,"venue":"Findings of ACL 2026","authors":["Qingyu Ren","Tianjun Pan","Xingzhou Chen","Xuhong Wang"],"authors_zh":"Qingyu Ren, Tianjun Pan, Xingzhou Chen, Xuhong Wang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"写作 reward model 需求级评测与强化学习训练论文","best_for_zh":"需要按显式写作约束评测或训练 reward model 的研究者。","confidence":"medium","one_line":["W²Bench and WRL provide fine-grained writing requirements and reward-model evaluation records.","以需求删除构造写作 reward model 排序评测，并用其训练 WRL。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.findings-acl.134/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Rainier-rq1/From_Coarse_to_Fine"}],"link_count":3,"sections":9},{"id":"pum-gain-based-prefix-evaluation-2026","title":"From Correctness to Utility: Gain-Based Prefix Evaluation for LLM Reasoning","year":2026,"venue":"arXiv 2026","authors":["Yuhang Zhou","Yixin Cao","Guangnan Ye"],"authors_zh":"Yuhang Zhou, Yixin Cao, Guangnan Ye","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","reward-modeling","search"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"推理前缀效用建模与偏好数据集论文","best_for_zh":"需要将推理前缀的后续求解价值转为成对偏好信号，并用于候选选择、搜索或强化学习的研究者。","confidence":"high","one_line":["PUM-MATH turns the improvement a reasoning prefix gives lightweight student solvers into 282,346 pairwise preference records for prefix utility modeling.","282,346 条推理前缀对，以学生模型求解增益形成偏好标签，服务 BoN、搜索与 RL。"],"why":"It labels prefixes by downstream solving utility rather than treating local step correctness as the target signal.","primary_link":"https://arxiv.org/abs/2606.07190","links":[{"key":"code","label":["Code","代码"],"url":"https://zhiqix.github.io/pum-project-page"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zhiqix/PUM-MATH"}],"link_count":4,"sections":9},{"id":"caution-reward-hacking-2026","title":"From Curiosity to Caution: Mitigating Reward Hacking for Best-of-N with Pessimism","year":2026,"venue":"ICLR 2026","authors":["Zhuohao Yu","Zhiwei Steven Wu","Adam Block"],"authors_zh":"Zhuohao Yu, Zhiwei Steven Wu, Adam Block","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"推理时奖励黑客缓解与不确定性选择","best_for_zh":"使用奖励模型做 Best-of-N 推理、需要控制过度优化和分布外选择的研究者。","confidence":"high","one_line":["Uses a learned uncertainty penalty to make Best-of-N reward selection pessimistic on atypical responses.","以典型回答上的预测误差估计不确定性，并在 Best-of-N 选择时惩罚可疑高分。"],"why":"It turns uncertainty about atypical responses into an inference-time safeguard against reward-model exploitation.","primary_link":"https://openreview.net/forum?id=EZn2TmBBfF","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=EZn2TmBBfF&name=supplementary_material"}],"link_count":3,"sections":9},{"id":"igds-interpretability-guided-selection-2026","title":"From Insight to Action: A Novel Framework for Interpretability-Guided Data Selection in Large Language Models","year":2026,"venue":"ACL 2026","authors":["Ling Shi","Xinwei Wu","Xiaohu Zhao","Hao Wang","Heng Liu","Yangyang Liu","Linlong Xu","Longyue Wang","Deyi Xiong","Weihua Luo"],"authors_zh":"Ling Shi、Xinwei Wu、Xiaohu Zhao 等（天津大学、阿里巴巴集团）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["mathematical-reasoning","summarization","machine-translation"],"tags":["data-selection","interpretability","sft","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"可解释性驱动的数据选择与监督微调研究","best_for_zh":"适合希望把模型内部机制转化为推理训练数据筛选规则的读者。","confidence":"high","one_line":["IGDS selects SFT data that activates causally validated internal task features.","该方法先用因果干预确认模型内部的任务特征，再选择最能激活这些特征的监督微调样本。"],"why":"It turns feature-level interpretability evidence into a concrete training-data selection rule.","primary_link":"https://aclanthology.org/2026.acl-long.283/","links":[],"link_count":2,"sections":9},{"id":"certified-geometry-proofs-survey-2026","title":"From Natural Language to Certified Geometry Proofs: A Survey of LLM-Augmented Verification and Neuro-Symbolic Theorem Proving","year":2026,"venue":"The Big Picture Workshop at ACL 2026","authors":["Ioannis Tzachristas","Georgios Tzachristas"],"authors_zh":"Ioannis Tzachristas、Georgios Tzachristas","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["evaluation"],"construction_layer":["reward_verifier_layer"],"domains":["geometry","theorem-proving","neuro-symbolic-reasoning"],"tags":["survey","geometry","formal-verification","theorem-proving","acl-2026"],"status":"verified","priority":"可读","paper_type_zh":"几何形式化证明综述","best_for_zh":"关注几何推理、自动形式化与证明检查的读者。","confidence":"high","one_line":["A survey of hybrid systems that turn geometry problems into checked proof artifacts.","综述把自然语言与图形几何题转化为可由符号工具或证明助手检查的证明。"],"why":"It helps readers distinguish a convincing geometry explanation from a proof accepted by an explicit verifier.","primary_link":"https://aclanthology.org/2026.bigpicture-main.1/","links":[],"link_count":1,"sections":9},{"id":"vlm-capability-curriculum-2026","title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","year":2026,"venue":"ICML 2026","authors":["Juncheng Wu","Hardy Chen","Haoqin Tu","Xianfeng Tang","Freda Shi","Hui Liu","Hanqing Lu","Cihang Xie","Yuyin Zhou"],"authors_zh":"Juncheng Wu、Hardy Chen、Haoqin Tu、Xianfeng Tang、Freda Shi、Hui Liu、Hanqing Lu、Cihang Xie、Yuyin Zhou","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["vision_language","visual_perception","visual_mathematics","mathematics"],"tags":["vlm-capcurriculum","capability-curriculum","difficulty-curriculum","repeated-sampling","pass-rate","correctness-vector","visual-perception","visual-reasoning","rlvr","grpo","answer-level-rollouts","open-data"],"status":"partial","priority":"必读","paper_type_zh":"视觉语言模型课程数据发布与 RLVR 构造方案","best_for_zh":"研究多模态 RLVR、重复采样、能力课程、难度课程及 rollout 数据审计的读者","confidence":"high","one_line":["VLM-CapCurriculum releases 32,736 staged RLVR problems with 16 extracted Qwen3-VL-8B answers, per-answer correctness flags, and a pass rate per row, enabling capability-first and within-stage difficulty curricula without releasing the underlying full reasoning completions.","VLM-CapCurriculum 发布 32,736 条仅含 train split 的分阶段 RLVR 题目；每条保留 16 个抽取答案、逐答案正确性和 pass_rate，但不发布原始 completion 或思维链。"],"why":"It turns repeated sampling into an auditable curriculum signal across perception, text reasoning, and visual reasoning, while making the central reuse boundary explicit: answer-level success vectors and images are public, but raw correct and rejected chains of thought, decontamination, and source-level license resolution are not.","primary_link":"https://openreview.net/forum?id=r7uOjvZdzO","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UCSC-VLAA/VLM-CapCurriculum"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/UCSC-VLAA/vlm-capcurriculum"},{"key":"project","label":["Project","项目主页"],"url":"https://ucsc-vlaa.github.io/VLM-CapCurriculum/"}],"link_count":6,"sections":9},{"id":"selection-refinement-instruction-data-2026","title":"From Selection to Refinement: Iterative Optimization for Instruction Data","year":2026,"venue":"ACL 2026","authors":["Hang Hu","Ziyan Liu","Rujie Wen","Ruihui Hou","Xueyan Wu","Mu Zhang","Jianxing Yu","Tong Ruan","Jingping Liu"],"authors_zh":"Hang Hu, Ziyan Liu, Rujie Wen, Ruihui Hou, Xueyan Wu, Mu Zhang, Jianxing Yu, Tong Ruan, Jingping Liu","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"可读","paper_type_zh":"训练数据选择或奖励设计论文","best_for_zh":"研究后训练数据消费与优化目标的读者。","confidence":"high","one_line":["Instruction data optimization identifies valuable records and iteratively refines weak ones before SFT.","指令数据优化先识别有价值的记录，再在 SFT 前迭代改写较弱样本。"],"why":"It makes the training-data decision inspectable.","primary_link":"https://aclanthology.org/2026.acl-long.1889/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/surihuhang/From-Selection-to-Refinement--Iterative-Optimization-for-Instruction-Data"}],"link_count":2,"sections":9},{"id":"rlvrr-reward-chain-open-generation-2026","title":"From Verifiable Dot to Reward Chain: Harnessing Verifiable Reference-based Rewards for Reinforcement Learning of Open-ended Generation","year":2026,"venue":"ICLR 2026","authors":["Yuxin Jiang","Yufei Wang","Qiyuan Zhang","Xingshan Zeng","Liangyou Li","Jierun Chen","Chaofan Tao","Haoli Bai","Lifeng Shang"],"authors_zh":"Yuxin Jiang, Yufei Wang, Qiyuan Zhang, Xingshan Zeng, Liangyou Li, Jierun Chen, Chaofan Tao, Haoli Bai, Lifeng Shang","tracks":["training_usage_optimization_objectives","audit_failure_contamination_verifier_attacks"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"必读","paper_type_zh":"训练数据选择或奖励设计论文","best_for_zh":"研究后训练数据消费与优化目标的读者。","confidence":"high","one_line":["RLVRR turns high-quality references into an ordered reward chain for RL on open-ended generation.","RLVRR 将高质量参考答案转换为有序奖励链，用于开放式生成的强化学习。"],"why":"It makes the training-data decision inspectable.","primary_link":"https://arxiv.org/abs/2601.18533","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YJiangcm/RLVRR"}],"link_count":3,"sections":9},{"id":"frontierscience-expert-science-2026","title":"FrontierScience: Evaluating AI's Ability to Perform Expert-Level Scientific Tasks","year":2026,"venue":"arXiv","authors":["Miles Wang","Robi Lin","Kat Hu","Joy Jiao","Neil Chowdhury","Ethan Chang","Tejal Patwardhan"],"authors_zh":"Miles Wang, Robi Lin, Kat Hu, Joy Jiao, Neil Chowdhury, Ethan Chang, Tejal Patwardhan","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["science","reasoning"],"tags":["science","rubric","expert_reasoning"],"status":"verified","priority":"必读","paper_type_zh":"专家级科学任务与过程 rubric 基准","best_for_zh":"需要评估科学推理、科研子任务或专家 Judge 的研究者。","confidence":"high","one_line":["FrontierScience pairs original Olympiad and PhD-level science tasks with granular research-task rubrics.","FrontierScience 以奥赛和博士级科学任务及细粒度 rubric，测量专家科学推理。"],"why":"It targets expert scientific work rather than saturated factual exams.","primary_link":"https://arxiv.org/abs/2601.21165","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openai/frontierscience"}],"link_count":3,"sections":9},{"id":"fs-researcher-2026","title":"FS-Researcher: Test-Time Scaling for Long-Horizon Research Tasks with File-System-Based Agents","year":2026,"venue":"ACL 2026","authors":["Chiwei Zhu","Benfeng Xu","Mingxuan Du","Shaohan Wang","Xiaorui Wang","Zhendong Mao","Yongdong Zhang"],"authors_zh":"Chiwei Zhu、Benfeng Xu、Mingxuan Du、Shaohan Wang、Xiaorui Wang、Zhendong Mao、Yongdong Zhang","tracks":["rollout_search_test_time_trace_data"],"source_role":["agent_environment","construction_recipe","scaling_study"],"verification_contract":["environmental","judgment_required"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","optimizer_scaffold"],"domains":["deep_research","web_agents","long_horizon_reasoning"],"tags":["deep-research","file-system","external-memory","long-horizon"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["FS-Researcher preserves source archives and notes in a shared file-system workspace for long-horizon research.","FS-Researcher 把资料存档与笔记保存在共享的文件系统工作区中，以支撑长程研究任务。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"It elevates durable workspace state to a Track 5 trace object.","primary_link":"https://aclanthology.org/2026.acl-long.288/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Ignoramus0817/FS-Researcher"}],"link_count":3,"sections":9},{"id":"gala-geometric-self-training-2026","title":"GALA: Geometric Data Selection with Strategic Prospecting for Large Language Model Self-Training","year":2026,"venue":"Findings of ACL 2026","authors":["Zhongwei Xie","Ruihao Liao","Zimo Wang","Chong Chen","Xian-Sheng Hua","Xiao Luo"],"authors_zh":"Zhongwei Xie, Ruihao Liao, Zimo Wang, Chong Chen, Xian-Sheng Hua, Xiao Luo","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["sft","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics"],"tags":["self-training","data-selection","trajectory-generation","dynamic-validation"],"status":"verified","priority":"可读","paper_type_zh":"自训练数据选择与轨迹构造研究","best_for_zh":"适合希望减少冗余合成推理、设计数据高效可验证轨迹自训练流程的读者。","confidence":"high","one_line":["GALA selects geometrically representative problems, prospectively generates diverse verified traces, and validates mini-batches before self-training.","GALA 选择几何上有代表性的题目，生成多样且可验证的推理轨迹，并在自训练前动态验证小批数据。"],"why":"It combines subset selection, trajectory construction, and mini-batch validation into the training-data consumer rather than treating them as separate offline filters.","primary_link":"https://aclanthology.org/2026.findings-acl.500/","links":[],"link_count":1,"sections":9},{"id":"general-process-reward-modeling-robotic-reinforcement-learning-2026","title":"General Process Reward Modeling for Robotic Reinforcement Learning","year":2026,"venue":"CVPR 2026","authors":["Huajie Tan","Sixiang Chen","Yijie Xu","Zixiao Wang","Yuheng Ji","Cheng Chi","Yaoxu Lyu","Zhongxia Zhao","Xiansheng Chen","Peterson Co","Shaoxuan Xie","Guocai Yao","Pengwei Wang","Zhongyuan Wang","Shanghang Zhang"],"authors_zh":"Huajie Tan、Sixiang Chen、Yijie Xu 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["robotic_manipulation","vision_language_reward_modeling","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"机器人操作的多视角过程奖励建模与进展监督数据论文","best_for_zh":"训练或评测细粒度机器人进展奖励、单样本任务适应和稠密强化学习。","confidence":"high","one_line":["Robo-Dopamine releases multi-view robot-state progress supervision for general process reward modeling and policy-invariant reward shaping in manipulation.","Robo-Dopamine 发布多视角机器人状态进展监督，用于通用过程奖励建模与操作任务中的策略不变奖励塑形。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://openaccess.thecvf.com/content/CVPR2026/html/Tan_General_Process_Reward_Modeling_for_Robotic_Reinforcement_Learning_CVPR_2026_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/FlagOpen/Robo-Dopamine"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tanhuajie2001/Robo-Dopamine-GRM-Dataset"},{"key":"project","label":["Project","项目主页"],"url":"https://robo-dopamine.github.io/"}],"link_count":5,"sections":9},{"id":"omniverifier-2026","title":"Generative Universal Verifier as Multimodal Meta-Reasoner","year":2026,"venue":"ICLR 2026 Oral","authors":["Xinchen Zhang","Xiaoying Zhang","Youbin Wu","Yanbin Cao","Renrui Zhang","Ruihang Chu","Ling Yang","Yujiu Yang","Guang Shi"],"authors_zh":"Xinchen Zhang, Xiaoying Zhang, Youbin Wu, Yanbin Cao, Renrui Zhang, Ruihang Chu, Ling Yang, Yujiu Yang, Guang Shi","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"多模态视觉验证器、评测基准与测试时改进论文","best_for_zh":"需要评测视觉生成、构建多模态 verifier 或审计模型 judge 的研究者。","confidence":"medium","one_line":["Open verifier release for multi-modal reasoning with reproducible evaluation assets.","以生成式通用 verifier 审计并迭代改进多模态视觉结果，配套发布 ViVerBench。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://openreview.net/forum?id=DM0Y0oL33T","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Cominclip/OmniVerifier"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/comin/ViVerBench"}],"link_count":4,"sections":9},{"id":"georc-geolocation-reasoning-chains-2026","title":"GeoRC: A Benchmark for Geolocation Reasoning Chains","year":2026,"venue":"ACL 2026","authors":["Mohit Talreja","Joshua Diao","Jim Thannikary James","Radu Casapu","Tejas Santanam","Ethan Mendes","Alan Ritter","Wei Xu","James Hays"],"authors_zh":"Mohit Talreja、Joshua Diao、Jim Thannikary James、Radu Casapu、Tejas Santanam、Ethan Mendes、Alan Ritter、Wei Xu、James Hays","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["trace_writing","release_audit"],"domains":["vision_language","geolocation","expert_reasoning"],"tags":["vlm","expert_data","llm_judge","geolocation"],"status":"verified","priority":"必读","paper_type_zh":"专家标注基准与 Judge 评测论文","best_for_zh":"关注多模态推理可审计性、专家参考答案和 LLM-as-a-Judge 的研究者。","confidence":"high","one_line":["GeoRC pairs 800 expert geolocation chains with 500 scenes to expose the gap between accurate location guesses and auditable visual reasoning.","GeoRC 以 500 个街景和 800 条冠军级玩家推理链，评测 VLM 的定位解释是否真正由视觉证据支撑。"],"why":"It makes expert visual-evidence reasoning a reusable evaluation surface rather than trusting a correct final location alone.","primary_link":"https://aclanthology.org/2026.acl-long.1883/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mohit-talreja/GeoRC"}],"link_count":4,"sections":9},{"id":"gitchameleon-2-0-evaluating-ai-code-generation-against-python-library-version-incompatib","title":"GitChameleon 2.0: Evaluating AI Code Generation Against Python Library Version Incompatibilities","year":2026,"venue":"ACL 2026","authors":["Diganta Misra","Nizar Islah","Victor May","Brice Rauby","Zihan Wang","Justine Gehring","Antonio Orvieto","Muawiz Sajjad Chaudhary","Eilif B. Muller","Irina Rish","Samira Ebrahimi Kahou","Massimo Caccia"],"authors_zh":"Diganta Misra, Nizar Islah, Victor May, Brice Rauby, Zihan Wang, Justine Gehring, Antonio Orvieto, Muawiz Sajjad Chaudhary, Eilif B. Muller, Irina Rish, Samira Ebrahimi Kahou, Massimo Caccia","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A 328-task benchmark with pinned Python-library versions and unit tests for version-compatible generation.","328 个带固定 Python 库版本和单测的补全任务，专门测版本兼容生成。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://aclanthology.org/2026.acl-long.2170/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mrcabbage972/GitChameleonBenchmark"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/cabbage972/GitChameleon-2.0"},{"key":"project","label":["Project","项目主页"],"url":"https://gitchameleon-2-0.github.io/"}],"link_count":6,"sections":9},{"id":"gittaskbench-2025","title":"GitTaskBench: A Benchmark for Code Agents Solving Real-World Tasks Through Code Repository Leveraging","year":2026,"venue":"AAAI 2026","authors":["Ziyi Ni","Huacan Wang","Shuo Zhang","Shuo Lu","Ziyang He","Wang You","Zhenheng Tang","Sen Hu","Bo Li","Chen Hu","Binxing Jiao","Daxin Jiang","Yuntao Du","Pin Lyu"],"authors_zh":"Ziyi Ni、Huacan Wang、Shuo Zhang、Shuo Lu、Ziyang He、Wang You、Zhenheng Tang、Sen Hu、Bo Li、Chen Hu、Binxing Jiao、Daxin Jiang、Yuntao Du、Pin Lyu","tracks":["environment_agent_trajectory_data","programmatically_verifiable_outcome_data"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer"],"domains":["software_engineering","repository_agents","multimodal_task_execution","image_processing","video_processing","speech_processing","physiological_signal_processing","security_privacy","web_scraping","office_document_processing"],"tags":["benchmark","code-agents","repository-agents","environment-interaction","multimodal","programmatic-grading","terminal-predicate","evaluation-only","official-data-release"],"status":"verified","priority":"必读","paper_type_zh":"代码仓库智能体评测基准与环境","best_for_zh":"研究代码智能体评测、环境反馈、终态验证器与可复现性审计的读者","confidence":"high","one_line":["GitTaskBench releases 54 repository-linked multimodal tasks with inputs and custom final-output graders for autonomous repository understanding, environment setup, execution, and delivery, while withholding a canonical action-observation rollout corpus.","GitTaskBench 发布 54 个与代码仓库绑定的多模态评测任务、输入和定制终态 grader，用于评测智能体理解仓库、配置环境、执行任务并交付结果的能力，但未发布规范化 action-observation rollout 语料库。"],"why":"It exposes the repository, dependency setup, final artifact, and task-specific terminal predicate as the evaluation data object, and its 65.04% environment-setup failure share shows why environment state and replay metadata are central to agent reasoning-data audits.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/download/40533/44494","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QuantaAlpha/GitTaskBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Nicole-Yi/GitTaskBench"},{"key":"project","label":["Project","项目主页"],"url":"https://gittaskbench.github.io/"}],"link_count":7,"sections":9},{"id":"glm-5-agentic-engineering-2026","title":"GLM-5: from Vibe Coding to Agentic Engineering","year":2026,"venue":"arXiv preprint","authors":["GLM-5 Team"],"authors_zh":"GLM-5 Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","agent_environment","construction_recipe","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward","trajectory_value"],"training_use":["sft","distillation","reward_modeling","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["software_engineering","terminal_agents","search_agents","tool_use","mathematical_reasoning","scientific_reasoning","coding","general_assistant"],"tags":["glm-5","zhipu","agentic-engineering","software-engineering","terminal-agent","search-agent","verifiable-environment","asynchronous-rl","slime","tito","grpo","outcome-reward","long-horizon","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型技术报告与智能体训练数据披露台账","best_for_zh":"研究长程智能体训练、可执行环境、异步 RL、软件工程数据与前沿模型数据审计的读者","confidence":"high","one_line":["GLM-5 combines a 28.5T-token base pipeline with multi-task SFT, reasoning/agentic/general RL, cross-stage distillation, and 10K+ executable SWE plus terminal/search environments, while releasing weights and slime infrastructure but not the training records, exact mixtures, budgets, or safety ledger.","GLM-5 披露了 28.5T base-token 路线、三类 SFT、四域 Reasoning RL 与全异步 Agentic RL，以及 10K+ SWE、数千 terminal 和 2M+ 网页 WKG 环境，但未开放训练记录、完整 reward、精确预算、污染审计或安全训练账本。"],"why":"The report makes the frontier agent-training stack unusually legible—from issue/PR and web sources through executable environments, token-correct asynchronous rollouts, judges and rewards to long-horizon evaluation—yet also shows how much provenance, calibration, licensing, and record-level auditability remains missing even for an open-weight release.","primary_link":"https://arxiv.org/abs/2602.15763","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zai-org/GLM-5"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zai-org/terminal-bench-2-verified"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zai-org/GLM-5"}],"link_count":5,"sections":9},{"id":"go-browse-2025","title":"Go-Browse: Training Web Agents with Structured Exploration","year":2026,"venue":"ICLR 2026","authors":["Apurva Gandhi","Graham Neubig"],"authors_zh":"Apurva Gandhi、Graham Neubig","tracks":["rollout_search_test_time_trace_data","environment_agent_trajectory_data"],"source_role":["construction_recipe","data_release","agent_environment"],"verification_contract":["judgment_required","environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["web_navigation","gui_agents","transactional_web_tasks"],"tags":["go-browse","web-agent","webarena","structured-exploration","graph-search","browsergym","trajectory-synthesis","vlm-judge","sft"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"medium","one_line":["Go-Browse releases raw and processed WebArena-clone trajectories plus a Qwen checkpoint from graph-frontier task discovery, VLM feasibility judgment, and prefixed/unprefixed browser rollouts.","Go-Browse 通过图前沿式任务发现、视觉语言模型可行性判定与带前缀及不带前缀的浏览器采样，发布 WebArena 克隆环境的原始与处理后轨迹以及一个 Qwen 检查点。"],"why":"It connects task sourcing, search-state reuse, environment interaction, judge-based filtering, and supervised agent training while leaving material release-rights and reproducibility gaps visible.","primary_link":"https://arxiv.org/abs/2506.03533","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ApGa/Go-Browse"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/apurvaga/go-browse-wa"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/apurvaga/go-browse-wa-qwen-7B"}],"link_count":8,"sections":9},{"id":"openai-gpt-5-3-codex-system-card-2026","title":"GPT-5.3-Codex System Card","year":2026,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["agent_training","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["coding","software_engineering","agentic_tool_use","safety"],"tags":["openai","gpt-5-3-codex","codex","system-card","frontier-report","data-disclosure-ledger","coding-agent","reinforcement-learning","safety"],"status":"partial","priority":"可读","paper_type_zh":"闭源前沿编码代理系统卡与数据披露台账","best_for_zh":"审计闭源编码代理的训练、反馈和部署披露边界的读者","confidence":"high","one_line":["GPT-5.3-Codex's system card discloses one RL safety intervention—simulated conflicting user edits with positive reinforcement for preserving them—while withholding the task, trajectory, reward, and training-environment ledger.","GPT-5.3-Codex 系统卡披露了一项 RL 安全干预：在 rollout 中模拟相互冲突的用户编辑，并对保留这些编辑给予正向强化；但任务、轨迹、奖励和训练环境台账仍未披露。"],"why":"It is a high-impact Track12 disclosure-boundary case: a concrete coding-agent RL behavior is named, but the evidence is still insufficient to reconstruct the data object, verification contract, or reproducible post-training recipe.","primary_link":"https://deploymentsafety.openai.com/gpt-5-3-codex/gpt-5-3-codex.pdf","links":[],"link_count":2,"sections":9},{"id":"openai-gpt-5-4-thinking-system-card-2026","title":"GPT-5.4 Thinking System Card","year":2026,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["agent_training","evaluation","audit","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","chain_of_thought","agentic_systems","computer_use","software_engineering","cybersecurity","safety","biosecurity","ai_self_improvement"],"tags":["openai","gpt-5-4-thinking","system-card","reasoning-reinforcement-learning","long-rollout","destructive-action","simulated-user-work","computer-use","confirmation-policy","dynamic-conversations","production-resampling","prompt-injection-overlap","cot-monitorability","cot-control","hidden-unit-tests","cyber-safety","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"Living 前沿推理系统卡与数据、监控及部署披露账本","best_for_zh":"审计长轨迹 agent 训练、CoT monitorability、评测 grader 和 cyber 部署监控边界的研究者","confidence":"high","one_line":["GPT-5.4 Thinking adds long-rollout self-reversion, configurable confirmation, prompt-injection, and cyber-safety training disclosures plus dynamic and production-like safety trajectories, CoT audits, and hidden-test agent evaluations, but not the training records, rewards, global split, or stable versioned artifacts.","GPT-5.4 Thinking 的 living system card 披露长 rollout 自回退并保护模拟用户工作、确认策略与 cyber 安全训练，同时加入动态/production-like 轨迹、CoT monitorability 和 programmatic agent 评测；其关键价值是分离训练、评测、monitor 与部署，而底层记录、reward 和全局 split 仍未公开。"],"why":"It is a high-value disclosure ledger because it exposes several concrete reasoning-agent data and feedback surfaces while making it possible to separate training interventions from resampled traffic, policy and task graders, CoT monitors, programmatic tests, and deployment controls.","primary_link":"https://deploymentsafety.openai.com/gpt-5-4-thinking/gpt-5-4-thinking.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-gpt-5-4/"}],"link_count":3,"sections":9},{"id":"gr-ben-general-reasoning-prm-benchmark-2026","title":"GR-Ben: A General Reasoning Benchmark for Evaluating Process Reward Models","year":2026,"venue":"arXiv 2026","authors":["Zhouhao Sun","Xuan Zhang","Xiao Ding","Bibo Cai","Li Du","Kai Xiong","Xinran Dai","Fei Zhang","Weidi Tang","Zhiyuan Kan","Yang Zhao","Bing Qin","Ting Liu"],"authors_zh":"Zhouhao Sun, Xuan Zhang, Xiao Ding, Bibo Cai, Li Du, Kai Xiong, Xinran Dai, Fei Zhang, Weidi Tang, Zhiyuan Kan, Yang Zhao, Bing Qin, Ting Liu","tracks":["preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["general-reasoning","process-reward-modeling","evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"跨领域过程奖励模型评测与错误类型标注基准论文","best_for_zh":"需要检验过程奖励模型在科学和逻辑等非数学领域定位错误步骤及区分错误类型能力的研究者。","confidence":"high","one_line":["GR-Ben evaluates process reward models on roughly 3.6K human-annotated science and logic traces with error positions, types, and reasons.","约3.6k条、9领域推理轨迹给出错误位置、类型与理由，直接测量 PRM 的跨域过程判断。"],"why":"It extends process-feedback evaluation past mathematics while preserving the annotations needed to analyse what kind of error a model misses.","primary_link":"https://arxiv.org/abs/2605.01203","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/spirit-moon-fly/GR-Ben"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GR-Ben/GR-Ben"}],"link_count":4,"sections":9},{"id":"grace-context-faithfulness-benchmark-2026","title":"GRACE: Step-Level Benchmark for Faithful Reasoning over Context","year":2026,"venue":"arXiv preprint","authors":["Hoang Pham","Dong Le","Anh Tuan Luu"],"authors_zh":"Hoang Pham、Dong Le、Anh Tuan Luu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["context-grounded-reasoning","faithfulness","process-supervision"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要构建上下文忠实性过程奖励、定位推理错误或评测证据约束推理的研究者。","confidence":"high","one_line":["GRACE supplies context-grounded reasoning traces with per-step faithfulness labels, error types, and explanations for training or testing process evaluators.","GRACE 提供带逐步忠实性标签、错误类型和解释的上下文推理轨迹，可训练或评测过程判别器。"],"why":"It separates context faithfulness from final-answer accuracy and offers both scalable consensus labels and a human-annotated hard test set.","primary_link":"https://arxiv.org/abs/2606.16151","links":[{"key":"data","label":["Data","数据"],"url":"https://anonymous.4open.science/r/grace-bench-14e8"}],"link_count":3,"sections":9},{"id":"gram-r2-self-training-reward-reasoning-2026","title":"GRAM-R²: Self-Training Generative Foundation Reward Models for Reward Reasoning","year":2026,"venue":"AAAI 2026","authors":["Chenglong Wang","Yongyu Mu","Hang Zhou","Yifu Huo","Ziming Zhu","Jiali Zeng","Murun Yang","Bei Li","Tong Xiao","Xiaoyang Hao","Chunliang Zhang","Fandong Meng","Jingbo Zhu"],"authors_zh":"Chenglong Wang, Yongyu Mu, Hang Zhou, Yifu Huo, Ziming Zhu, Jiali Zeng, Murun Yang, Bei Li, Tong Xiao, Xiaoyang Hao, Chunliang Zhang, Fandong Meng, Jingbo Zhu","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["preference-modeling","reward-reasoning","alignment"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"自训练生成式基础奖励模型与奖励推理数据论文","best_for_zh":"需要将无理由偏好数据和无标注数据扩展为带比较理由的奖励模型训练语料的研究者。","confidence":"high","one_line":["GRAM-R² self-trains a generative foundation reward model by turning labeled and unlabeled preference pairs into rationale-backed comparison supervision.","28.5GB 公开训练集；百万级偏好对经自训练生成“反馈—比较—结论—A/B”监督，用于奖励推理。"],"why":"It exposes an explicit feedback–comparison–conclusion format for scaling reward reasoning beyond manually written rationales.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40626","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/wangclnlp/GRAM-RR-TrainingData"},{"key":"project","label":["Project","项目主页"],"url":"https://wangclnlp.github.io/wangchenglong.github.io/"}],"link_count":5,"sections":9},{"id":"gui-libra-2026","title":"GUI-Libra: Training Native GUI Agents to Reason and Act with Action-aware Supervision and Partially Verifiable RL","year":2026,"venue":"arXiv preprint (2026)","authors":["Rui Yang","Qianhui Wu","Zhaoyang Wang","Hanyang Chen","Ke Yang","Hao Cheng","Huaxiu Yao","Baolin Peng","Huan Zhang","Jianfeng Gao","Tong Zhang"],"authors_zh":"Rui Yang, Qianhui Wu, Zhaoyang Wang, Hanyang Chen, Ke Yang, Hao Cheng, Huaxiu Yao, Baolin Peng, Huan Zhang, Jianfeng Gao, Tong Zhang","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","agent_training","rlvr"],"construction_layer":["trace_writing","reward_verifier_layer"],"domains":["gui-agents","multimodal-reasoning"],"tags":["gui-reasoning-data","action-aware-sft","partially-verifiable-rl"],"status":"verified","priority":"可读","paper_type_zh":"GUI 推理数据集与后训练配方","best_for_zh":"适合构造原生 GUI agent 推理与动作监督数据的研究者。","confidence":"high","one_line":["GUI-Libra turns public GUI trajectories into 81K action-aligned reasoning steps and trains native agents with action-aware SFT and conservative RL under ambiguous step rewards.","GUI-Libra 将公开 GUI 轨迹整理为 81K 条动作对齐推理记录，并用 action-aware SFT 与保守 RL 处理逐步奖励存在歧义的问题。"],"why":"It exposes the record schema, filters, and reward ambiguity needed to judge whether offline GUI supervision will transfer to online task completion.","primary_link":"https://arxiv.org/abs/2602.22190","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GUI-Libra/GUI-Libra"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/GUI-Libra"},{"key":"project","label":["Project","项目主页"],"url":"https://gui-libra.github.io"}],"link_count":4,"sections":9},{"id":"guided-gut-intrinsic-confidence-2026","title":"Guided by Gut: Efficient Test-Time Scaling with Reinforced Intrinsic Confidence","year":2026,"venue":"ACL 2026 Long Papers","authors":["Amirhosein Ghasemabadi","Keith G. Mills","Baochun Li","Di Niu"],"authors_zh":"Amirhosein Ghasemabadi（阿尔伯塔大学）、Keith G. Mills（路易斯安那州立大学）、Baochun Li（多伦多大学）、Di Niu（阿尔伯塔大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","reinforcement_learning","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code-generation"],"tags":["test-time-scaling","intrinsic-confidence","tree-search","grpo","efficient-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"高效置信度引导测试时扩展研究","best_for_zh":"希望以低显存部署推理搜索、并需判断模型内部不确定性能否替代独立验证器的读者。","confidence":"high","one_line":["Guided by Gut calibrates a model’s own step confidence with GRPO and uses it to guide lightweight tree search without a process reward model.","Guided by Gut 先用 GRPO 校准模型自身的步骤置信度，再以它引导轻量树搜索，无需在推理时调用过程奖励模型。"],"why":"It makes confidence calibration a deployment decision: a smaller model can spend inference compute on promising branches instead of paying to score every branch with an external model.","primary_link":"https://aclanthology.org/2026.acl-long.739/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GAIR/LIMO"}],"link_count":3,"sections":9},{"id":"hack-verifiable-environments-2026","title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","year":2026,"venue":"arXiv","authors":["Amit Roth","Ankur Samanta","Matan Halevy","Yoav Levine","Yonathan Efroni"],"authors_zh":"Amit Roth, Ankur Samanta, Matan Halevy, Yoav Levine, Yonathan Efroni","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"智能体奖励投机与确定性环境审计论文","best_for_zh":"需要评测智能体奖励投机、环境漏洞或自动化对齐风险的研究者。","confidence":"high","one_line":["deterministic reward-hacking testbed based on TextArena","以植入捷径和确定性监控，将智能体奖励投机从事后判断转为可规模化测量。"],"why":"It separates task success from deterministic evidence that success used an unintended shortcut.","primary_link":"https://arxiv.org/abs/2605.20744","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MajoRoth/hack-verifiable-environments"}],"link_count":2,"sections":9},{"id":"had-hallucination-detection-taxonomy-2026","title":"HAD: HAllucination Detection Language Models Based on a Comprehensive Hallucination Taxonomy","year":2026,"venue":"ACL 2026 Industry Track","authors":["Fan Xu","Xinyu Hu","Zhenghan Yu","Li Lin","Xu Zhang","Yang Zhang","Wei Zhou","Jinjie Gu","Xiaojun Wan"],"authors_zh":"Fan Xu、Xinyu Hu、Zhenghan Yu、Li Lin、Xu Zhang、Yang Zhang、Wei Zhou、Jinjie Gu、Xiaojun Wan","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","summarization"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"可读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["HAD is the paper's released feedback or evaluation resource.","建立跨 NLG 任务的 11 类幻觉 taxonomy、约 90K 训练实例及 2,248 条人工标注 HADTest，输出错误类别、span 和修正建议。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2026.acl-industry.11/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pku0xff/HAD"}],"link_count":3,"sections":9},{"id":"halluguard-evidence-grounded-small-reasoning-models-to-mitigate-hallucinations-in-retrieval-augmente-2026","title":"HalluGuard: Evidence-Grounded Small Reasoning Models to Mitigate Hallucinations in Retrieval-Augmented Generation","year":2026,"venue":"Findings of ACL 2026","authors":["Loris Bergeron","Ioana Buhnila","Jérôme François","Radu State"],"authors_zh":"Loris Bergeron、Ioana Buhnila、Jérôme François、Radu State","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"检索增强生成幻觉检测的偏好数据与奖励模型论文","best_for_zh":"研究检索增强生成可靠性、奖励建模或偏好优化的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","HalluGuard 以证据支撑的偏好元组训练小型推理模型，用于识别检索增强生成中的幻觉。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://aclanthology.org/2026.findings-acl.835/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lrsbrgrn/HalluGuard-Preferences-76k"}],"link_count":2,"sections":9},{"id":"hard2verify-step-level-verification-benchmark-2026","title":"Hard2Verify: A Step-Level Verification Benchmark for Open-Ended Frontier Math","year":2026,"venue":"ACL 2026","authors":["Shrey Pandit","Austin Xu","Xuan-Phi Nguyen","Yifei Ming","Caiming Xiong","Shafiq Joty"],"authors_zh":"Shrey Pandit, Austin Xu, Xuan-Phi Nguyen, Yifei Ming, Caiming Xiong, Shafiq Joty","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-reward-modeling","evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测开放式高难数学证明中逐步核验、首错定位与验证器泛化能力的研究者。","confidence":"high","one_line":["Hard2Verify provides over 500 hours of human step-level annotations for testing verifiers on recent, open-ended frontier mathematics.","Hard2Verify 以 500 余小时人工标注开放前沿数学证明的步骤正确性与首错位置，检验高难过程验证上限。"],"why":"It supplies a deliberately contamination-protected human evaluation surface for verification beyond short-answer mathematics.","primary_link":"https://aclanthology.org/2026.acl-long.1031/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SalesforceAIResearch/Hard2Verify"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Salesforce/Hard2Verify"}],"link_count":5,"sections":9},{"id":"harm-hate-aware-reward-model-2026","title":"HARM: Learning Hate-Aware Reward Model for Evaluating Natural Language Explanations of Offensive Content","year":2026,"venue":"Findings of EACL 2026","authors":["Lorenzo Puppi Vecchi","Alceu de Souza Britto Jr","Emerson Cabrera Paraiso","Rafael Menelau Cruz"],"authors_zh":"Lorenzo Puppi Vecchi, Alceu de Souza Britto Jr, Emerson Cabrera Paraiso, Rafael Menelau Cruz","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"判断、rubric 或领域专家数据论文","best_for_zh":"需要构建或审计 LLM-as-a-judge、奖励模型与专家评测数据的研究者。","confidence":"medium","one_line":["A human-validated explanation dataset and reward-model setup for audit of offensive-content explanations.","以分层人工语境重校奖励信号，避免仇恨言论解释因必要的冒犯措辞而被错误降分。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.findings-eacl.230/","links":[],"link_count":1,"sections":9},{"id":"harness-bench-harness-effects-2026","title":"Harness-Bench: Measuring Harness Effects across Models in Realistic Agent Workflows","year":2026,"venue":"arXiv preprint / Diagnostic Benchmark 2026","authors":["Yilun Yao","Xinyu Tan","Chao-Hsuan Liu","Yaoming Li","Zhengyang Wang","Wenhan Yu","Zhewen Tan","Yuxuan Tian","Guangxiang Zhao","Lin Sun","Xiangzheng Zhang","Tong Yang"],"authors_zh":"Yilun Yao 等（Peking University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["agent-harnesses","workflow-trajectories","benchmark-diagnosis"],"tags":["agent_environment","trajectory_data","agent-harnesses","workflow-trajectories","benchmark-diagnosis"],"status":"verified","priority":"可读","paper_type_zh":"arXiv / Diagnostic Benchmark 2026 的 agent workflow harness benchmark","best_for_zh":"关注软件工程智能体、环境轨迹数据、执行反馈和评测 harness 的研究者。","confidence":"high","one_line":["Harness-Bench studies how harness choices change measured agent workflow performance across realistic tasks and models.","Harness-Bench 衡量 harness 选择如何改变真实智能体工作流中的模型表现。"],"why":"It makes harness configuration a first-class audit object for agent workflow evaluation.","primary_link":"https://arxiv.org/abs/2605.27922","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Qihoo360/harness-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://www.harness-bench.ai/"}],"link_count":4,"sections":9},{"id":"heteroskedastic-budgeted-verification-2026","title":"Heteroskedastic Signals in Budgeted LLM Verification: Structural Heterogeneity Limits Optimization Gains","year":2026,"venue":"arXiv preprint","authors":["Jinlong Yang"],"authors_zh":"Jinlong Yang（机构：西北工业大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code"],"tags":["test-time-compute","budgeted-verification","uncertainty","adaptive-allocation","cost-stratification"],"status":"verified","priority":"可读","paper_type_zh":"预算化验证与测试时分配分析","best_for_zh":"诊断基于不确定性的验证路由，或设计预算感知测试时计算策略的读者。","confidence":"high","one_line":["The paper shows that stratifying a limited verification budget by cost can outperform stronger globally shared adaptive policies when uncertainty signals are not comparable across strata.","当不确定性信号在不同成本层中不可比较时，按成本分层分配有限验证预算可优于更强的全局自适应策略。"],"why":"It gives a negative result with a practical remedy: optimizing a stronger global allocator cannot fix feedback signals whose reliability differs systematically by cost regime.","primary_link":"https://arxiv.org/abs/2606.15841","links":[],"link_count":2,"sections":9},{"id":"step-hidden-state-pruning-2026","title":"Hidden States as Early Signals: Step-level Trace Evaluation and Pruning for Efficient Test-Time Scaling","year":2026,"venue":"Findings of ACL 2026","authors":["Zhixiang Liang","Beichen Huang","Zheng Wang","Minjia Zhang"],"authors_zh":"Zhixiang Liang、Beichen Huang、Zheng Wang、Minjia Zhang（机构：伊利诺伊大学厄巴纳—香槟分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-scaling","trace-pruning","hidden-states","kv-cache","latency"],"status":"verified","priority":"必读","paper_type_zh":"高效并行测试时扩展研究","best_for_zh":"部署自一致性类推理、且 GPU 内存压力与排队而非单纯词元数决定延迟的读者。","confidence":"high","one_line":["STEP prunes low-promise reasoning traces using step-end hidden states when KV-cache pressure rises, reducing parallel test-time latency without sacrificing accuracy.","STEP 在 KV 缓存压力上升时，根据步骤末端隐藏状态剪去低前景推理轨迹，以降低并行测试时延迟而不牺牲准确率。"],"why":"It connects trace-quality signals to an actual serving-system trigger, showing that compute allocation must consider KV-cache scheduling as well as model reasoning.","primary_link":"https://aclanthology.org/2026.findings-acl.1336/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Supercomputing-System-AI-Lab/STEP"}],"link_count":3,"sections":9},{"id":"hiersva-a-data-synthesis-pipeline-dataset-and-benchmark-for-llm-driven-hierarchical-hard","title":"HierSVA: A Data Synthesis Pipeline, Dataset, and Benchmark for LLM-Driven Hierarchical Hardware Formal Verification","year":2026,"venue":"arXiv","authors":["Maohua Nie","Jiang Zhu","Jingqun Zhang","Zhichen Zeng","Jiayi Wang","Sibo Zhang","Jialin Wang","C. -J. Richard Shi"],"authors_zh":"Maohua Nie, Jiang Zhu, Jingqun Zhang, Zhichen Zeng, Jiayi Wang, Sibo Zhang, Jialin Wang, C. -J. Richard Shi","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A six-axis SVA benchmark with 342 hierarchical RTL modules and 28 deep bug pairs.","342 个层级 RTL 模块、28 组深层 bug pair 与六维 SVA 验证 benchmark。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2606.13706","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HierSVAAnon/HierSVACodeAndArtifacts"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AnonymousHierSVA/HierSVA"}],"link_count":3,"sections":9},{"id":"hippocamp-contextual-agents-pc-2026","title":"HippoCamp: Benchmarking Contextual Agents on Personal Computers","year":2026,"venue":"ECCV 2026","authors":["Zhe Yang","Shulin Tian","Kairui Hu","Shuai Liu","Hoang-Nhat Nguyen","Yichi Zhang","Zujin Guo","Mengying Yu","Zinan Zhang","Jingkang Yang","Chen Change Loy","Ziwei Liu"],"authors_zh":"Zhe Yang、Shulin Tian、Kairui Hu、Shuai Liu、Hoang-Nhat Nguyen、Yichi Zhang、Zujin Guo、Mengying Yu、Zinan Zhang、Jingkang Yang、Chen Change Loy、Ziwei Liu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["contextual-agents","personal-computing","stepwise-diagnosis"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究个人计算、长程多模态检索和智能体过程诊断的读者。","confidence":"high","one_line":["HippoCamp evaluates contextual agents over realistic personal file systems with evidence-grounded rationales and step-wise failure labels.","HippoCamp 在真实个人文件系统上评测上下文智能体，并提供证据链与逐步失败诊断标注。"],"why":"It pairs device-scale multimodal context with structured evidence and process annotations for diagnosing agent failure stages.","primary_link":"https://arxiv.org/abs/2604.01221","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HippoCamp-AI/HippoCamp"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MMMem-org/HippoCamp"},{"key":"project","label":["Project","项目主页"],"url":"https://hippocamp-ai.github.io/"}],"link_count":5,"sections":9},{"id":"honeybee-vl-reasoning-2026","title":"HoneyBee: Data Recipes for Vision-Language Reasoners","year":2026,"venue":"CVPR 2026","authors":["Hritik Bansal","Devendra Singh Sachan","Kai-Wei Chang","Aditya Grover","Gargi Ghosh","Wen-tau Yih","Ramakanth Pasunuru"],"authors_zh":"Hritik Bansal、Devendra Singh Sachan、Kai-Wei Chang、Aditya Grover、Gargi Ghosh、Wen-tau Yih、Ramakanth Pasunuru","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","scaling_report","release_audit"],"domains":["multimodal","visual_reasoning","mathematics","science","text_reasoning"],"tags":["honeybee","vision-language-reasoning","multimodal-cot","llama4-scout","data-scaling","caption-and-solve","pointer-based-release","same-teacher-majority"],"status":"verified","priority":"必读","paper_type_zh":"视觉语言推理数据配方与开放数据发布","best_for_zh":"关注多模态 CoT、教师蒸馏、数据缩放实验、来源指针和许可审计的研究者","confidence":"high","one_line":["HoneyBee reports 2.480M caption-plus-solution and text-only SFT records built with Llama-4 Scout around 28K images and 350K questions; the live release is pointer-based, partially materialized, and governed by layered licenses.","HoneyBee 报告了 248.0 万条由 Llama-4 Scout 构造的图文与纯文本 SFT 记录，覆盖 2.8 万张图像和 35 万个问题；公开版本以来源指针为主，当前仅部分物化，并受多层许可约束。"],"why":"It turns context choice, auxiliary captions, image diversity, questions per image, and traces per question into separately tested construction variables, giving the open-release recipe track a concrete multimodal scaling study with visible audit boundaries.","primary_link":"https://openaccess.thecvf.com/content/CVPR2026/html/Bansal_HoneyBee_Data_Recipes_for_Vision-Language_Reasoners_CVPR_2026_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/HoneyBee_VLM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/HoneyBee"}],"link_count":4,"sections":9},{"id":"reasoning-evolves-chess-2026","title":"How Reasoning Evolves from Post-Training Data: An Empirical Study Using Chess","year":2026,"venue":"ICML 2026","authors":["Lucas Dionisopoulos","Nicklas Majamaki","Prithviraj Ammanabrolu"],"authors_zh":"Lucas Dionisopoulos、Nicklas Majamaki、Prithviraj Ammanabrolu","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["chess","reasoning"],"tags":["track5","chess","rejection-sampling","verbalized-alpha-beta","search-traces","stockfish","sft","rlvr"],"status":"verified","priority":"必读","paper_type_zh":"国际象棋推理数据发布与后训练实证研究","best_for_zh":"需要审计 rollout、搜索轨迹、程序化验证、SFT/RL 数据谱系及其复现边界的读者","confidence":"high","one_line":["Releases a 120M-token chess SFT dataset and code spanning filtered Llama 4 Maverick responses, programmatic alpha-beta-style search traces, and engine-derived best-move and best-line targets, then studies their downstream RL behavior.","该工作发布一个约 1.20 亿 token 的国际象棋 SFT 数据集和代码；数据涵盖经筛选的 Llama 4 Maverick 回答、程序化的类 alpha-beta 搜索轨迹，以及引擎生成的最佳走法和最佳线路目标，并研究其下游 RL 行为。"],"why":"It exposes concrete Track 5 data objects—accepted teacher responses and verbalized, engine-valued search paths—while documenting the selection contract and the boundary that rejected pools, sampling parameters, and raw-position provenance are still incomplete.","primary_link":"https://arxiv.org/abs/2604.05134","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lucasdino/lang-chess"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lucasdino/chess-reasoning-data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/lucasdino/llm-chess"}],"link_count":6,"sections":9},{"id":"howtobench-tree-of-writing-2026","title":"HoWToBench: Holistic Evaluation for LLM’s Capability in Human-level Writing using Tree of Writing","year":2026,"venue":"ACL 2026","authors":["Andrew Zhuoer Feng","Cunxiang Wang","Yu Luo","Lin Fan","Yilin Zhou","Zikang Wang","Xiaotao Gu","Jie Tang","Hongning Wang","Minlie Huang"],"authors_zh":"Andrew Zhuoer Feng、Cunxiang Wang、Yu Luo、Lin Fan、Yilin Zhou、Zikang Wang、Xiaotao Gu、Jie Tang、Hongning Wang、Minlie Huang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","filtering_curation","release_audit"],"domains":["writing","llm_judge","chinese"],"tags":["rubric","writing","llm_judge","chinese"],"status":"verified","priority":"可读","paper_type_zh":"Rubric 基准与评测方法论文","best_for_zh":"研究开放式写作评测、可解释 Judge 或中文生成质量的研究者。","confidence":"high","one_line":["HoWToBench evaluates 1,302 Chinese writing instructions with a transparent Tree-of-Writing rubric that makes score aggregation explicit.","HoWToBench 用 1,302 条中文写作指令和显式加权的 Tree-of-Writing rubric，评估开放式长文写作质量。"],"why":"It addresses opaque rubric aggregation and tests robustness to disturbance rather than equating quality with fluent long text.","primary_link":"https://aclanthology.org/2026.acl-long.317/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ZhuoerFeng/ACL2026-Tree-of-Writing"}],"link_count":3,"sections":9},{"id":"hubble-memorization-suite-2026","title":"Hubble: a Model Suite to Advance the Study of LLM Memorization","year":2026,"venue":"ICLR 2026 Oral","authors":["Johnny Tian-Zheng Wei","Ameya Godbole","Mohammad Aflah Khan","Ryan Wang","Xiaoyuan Zhu","James Flemings","Nitya Kashyap","Krishna P. Gummadi","Willie Neiswanger","Robin Jia"],"authors_zh":"Johnny Tian-Zheng Wei, Ameya Godbole, Mohammad Aflah Khan, Ryan Wang, Xiaoyuan Zhu, James Flemings, Nitya Kashyap, Krishna P. Gummadi, Willie Neiswanger, Robin Jia","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"大模型记忆、隐私与测试集污染的受控模型套件","best_for_zh":"研究成员推断、模型遗忘、隐私泄漏或测试集污染的研究者。","confidence":"medium","one_line":["Controlled paired LLMs with injected text and test sets make memorization and contamination experiments reproducible.","以可控文本插入的开源配对模型，将记忆与污染研究变为可复现的因果实验。"],"why":"It supplies an inspectable testbed for measuring when sensitive text is memorized and for evaluating mitigations.","primary_link":"https://openreview.net/forum?id=ZfdnZhOP0k","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=ZfdnZhOP0k&name=supplementary_material"}],"link_count":2,"sections":9},{"id":"humanitys-last-exam-expert-knowledge-2026","title":"Humanity's Last Exam","year":2026,"venue":"Nature","authors":["Long Phan","Alice Gatti","Ziwen Han"],"authors_zh":"Long Phan、Alice Gatti、Ziwen Han","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"可读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["Humanity's Last Exam uses expert-written multimodal questions to probe the frontier of human knowledge.","由跨学科专家题目构成的多模态极难基准，用于追踪前沿模型的知识边界。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://arxiv.org/abs/2501.14249","links":[{"key":"data","label":["Data","数据"],"url":"https://lastexam.ai/"}],"link_count":2,"sections":9},{"id":"hwe-bench-hardware-bug-repair-2026","title":"HWE-Bench: Benchmarking LLM Agents on Real-World Hardware Bug Repair Tasks","year":2026,"venue":"arXiv","authors":["Fan Cui","Hongyuan Hou","Zizhang Luo","Chenyun Yin","Yun Liang"],"authors_zh":"Fan Cui, Hongyuan Hou, Zizhang Luo, Chenyun Yin, Yun Liang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code","software-engineering"],"tags":["hardware","hdl","bug-repair","simulation","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"带原生仿真验证的仓库级硬件修复基准","best_for_zh":"研究 HDL 代理、硬件修复和仿真式奖励的研究者。","confidence":"high","one_line":["HWE-Bench evaluates hardware agents on real repository bugs using native simulation and regression tests.","HWE-Bench 在真实硬件仓库上以原生仿真和回归测试评估代理修复历史 bug 的能力。"],"why":"It brings execution-verified, repository-level repair evaluation to RTL and hardware-software artifacts.","primary_link":"https://arxiv.org/abs/2604.14709","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/henryen/hwe-bench"}],"link_count":3,"sections":9},{"id":"if-rewardbench-instruction-following-2026","title":"IF-RewardBench: Benchmarking Judge Models for Instruction-Following Evaluation","year":2026,"venue":"ACL 2026","authors":["Bosi Wen","Yilin Niu","Cunxiang Wang","Xiaoying Ling","Ying Zhang","Pei Ke","Hongning Wang","Minlie Huang"],"authors_zh":"Bosi Wen、Yilin Niu、Cunxiang Wang、Xiaoying Ling、Ying Zhang、Pei Ke、Hongning Wang、Minlie Huang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["IF-RewardBench uses expert-validated constraint labels to form preference graphs that test listwise ranking by instruction-following judges.","IF-RewardBench 用专家核验的约束标签构建偏好图，测试评判模型对指令遵循回答的列表排序能力。"],"why":"IF-RewardBench uses expert-validated constraint labels to form preference graphs that test listwise ranking by instruction-following judges.","primary_link":"https://aclanthology.org/2026.acl-long.1092/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thu-coai/IF-RewardBench"}],"link_count":3,"sections":9},{"id":"semantic-equivalence-self-play-2026","title":"Improving LLM Code Reasoning via Semantic Equivalence Self-Play with Formal Verification","year":2026,"venue":"Findings of ACL 2026","authors":["Poon Tsz Nok","Antonio Valerio Miceli Barone"],"authors_zh":"Poon Tsz Nok、Antonio Valerio Miceli Barone","tracks":["data_construction_open_release_recipes","process_trace_supervision_data"],"source_role":["construction_recipe","data_release","verifier_reward","agent_environment","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level","scalar_reward","full_episode"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["code","haskell","formal_verification","semantic_equivalence"],"tags":["semantic-equivalence","self-play","liquid-haskell","execution-counterexample","open-data","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"数据构造与开放发布","best_for_zh":"关注代码推理数据、混合验证器、自博弈课程与发布审计的研究者","confidence":"high","one_line":["Releases 28,253 runnable Haskell references and specifies proof/counterexample self-play, but not the central verified interaction corpus or adapters.","公开 28,253 条可运行 Haskell 参考程序，并给出形式证明/执行反例自博弈方案，但未发布核心验证交互语料与适配器。"],"why":"Shows how formal positive evidence and executable negative witnesses can be combined in a data curriculum, while making the verification-yield and reproducibility trade-offs measurable.","primary_link":"https://aclanthology.org/2026.findings-acl.1615/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Baki-0501/llm-self-play-liquidhaskell"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Trevor0501/OpInstruct-HSx"}],"link_count":6,"sections":9},{"id":"closed-source-nli-ood-selection-2026","title":"Improving the OOD Performance of Closed-Source LLMs on NLI Through Strategic Data Selection","year":2026,"venue":"Findings of EACL 2026","authors":["Joe Stacey","Lisa Alazraki","Aran Ubhi","Beyza Ermis","Aaron Mueller","Marek Rei"],"authors_zh":"Joe Stacey, Lisa Alazraki, Aran Ubhi, Beyza Ermis, Aaron Mueller, Marek Rei","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["natural_language_inference","robustness","closed_source_finetuning"],"tags":["data-selection","synthetic-data","ood-robustness","nli","closed-source-llm"],"status":"verified","priority":"可读","paper_type_zh":"固定预算、面向鲁棒性的数据选择研究","best_for_zh":"适合使用 API 微调、且必须通过改变训练集合而非目标函数或优化器来改善分布鲁棒性的读者。","confidence":"high","one_line":["This study selects or replaces NLI fine-tuning records under a fixed API-training budget to improve OOD robustness of closed-source LLMs.","本研究在固定 API 训练预算下选择或替换 NLI 微调记录，以提高闭源大语言模型的 OOD 鲁棒性。"],"why":"It shows that training-set composition can be the only controllable optimization lever for closed-source fine-tuning and measures the resulting OOD trade-offs.","primary_link":"https://aclanthology.org/2026.findings-eacl.286/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/joestacey/LLM_robustness_NLI"}],"link_count":3,"sections":9},{"id":"deepverifier-inference-time-verification-2026","title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","year":2026,"venue":"Findings of ACL 2026","authors":["Yuxuan Wan","Tianqing Fang","Zaitang Li","Yintong Huo","Wenxuan Wang","Haitao Mi","Dong Yu","Michael R. Lyu"],"authors_zh":"Yuxuan Wan、Tianqing Fang、Zaitang Li、Yintong Huo、Wenxuan Wang、Haitao Mi、Dong Yu、Michael R. Lyu（机构：香港中文大学、腾讯 AI Lab、新加坡管理大学、中国人民大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["agentic-reasoning","information-seeking"],"tags":["test-time-compute","agent-verification","rubric-guided-feedback","iterative-refinement","deep-research"],"status":"verified","priority":"必读","paper_type_zh":"智能体验证与测试时扩展研究","best_for_zh":"构建需要把额外推理计算投入循证自我纠错的联网智能体的读者。","confidence":"high","one_line":["DeepVerifier converts agent failure rubrics into targeted test-time checks and feedback loops that improve deep-research-agent answers through repeated verification and retry.","DeepVerifier 将智能体失败量表转为有针对性的测试时检查与反馈循环，通过反复验证和重试提升深度研究智能体的答案。"],"why":"It operationalizes verification as a repeatable inference loop rather than a final judge call, making extra agent compute target diagnosed failure points.","primary_link":"https://arxiv.org/abs/2601.15808","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yxwan123/DeepVerifier"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/Tencent/CognitiveKernel-Pro"}],"link_count":4,"sections":9},{"id":"infinite-problem-generator-2026","title":"Infinite Problem Generator: Verifiably Scaling Physics Reasoning Data with Agentic Workflows","year":2026,"venue":"arXiv preprint (2026)","authors":["Aditya Sharan","Sriram Hebbale","Dhruv Kumar"],"authors_zh":"Aditya Sharan, Sriram Hebbale, Dhruv Kumar","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer"],"domains":["physics-reasoning"],"tags":["physics-data-synthesis","executable-verification","formula-as-code"],"status":"verified","priority":"可读","paper_type_zh":"可执行物理数据生成器与数据集","best_for_zh":"适合构造可程序检查的物理训练题或评测集的研究者。","confidence":"high","one_line":["Infinite Problem Generator expands textbook physics seeds into 1,335 problem-code records whose predefined formulas and executed numeric solutions make generation verifiable.","Infinite Problem Generator 将教材物理题扩展为 1,335 条题目与代码记录，并用预定义公式和实际执行的数值解验证生成结果。"],"why":"It provides an operational recipe and record schema for generating physics data with stronger checks than answer-only teacher generation.","primary_link":"https://arxiv.org/abs/2603.14486","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/er-ads/ProblemGenerationAgent"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/erads/ClassicalMechanicsV1"},{"key":"project","label":["Project","项目主页"],"url":"https://er-ads.github.io/ProblemGenerationAgent/Physics_Evaluation_Report.html"}],"link_count":4,"sections":9},{"id":"infoes-online-rlhf-selection-2026","title":"Influence-based Online Experience Selection for Effective RLHF","year":2026,"venue":"ACL 2026","authors":["Yifan Gong","Jing Yao","Xiting Wang","Xunlong Wang","Xiaoyuan Yi","Xing Xie"],"authors_zh":"Yifan Gong, Jing Yao, Xiting Wang, Xunlong Wang, Xiaoyuan Yi, Xing Xie","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["agent_training","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","rlhf","online_rl"],"tags":["rlhf","online-data-selection","influence-functions","policy-optimization","alignment"],"status":"verified","priority":"必读","paper_type_zh":"目标感知的在线 RLHF 经验选择研究","best_for_zh":"适合构建 RLHF 系统、需要删除无关或奖励受污染 rollout，同时保留标准强化学习优化器兼容性的读者。","confidence":"high","one_line":["InfOES filters online RLHF experiences by their estimated influence on an objective-specific validation return before every policy update.","InfOES 在每次策略更新前，按在线 RLHF 经验对目标特定验证回报的估计影响筛选经验。"],"why":"It directly connects each training experience to the gradient of the desired alignment objective instead of relying on generic difficulty or length heuristics.","primary_link":"https://aclanthology.org/2026.acl-long.2206/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/paraGONG/InfOES"}],"link_count":3,"sections":9},{"id":"instructdiff-contrastive-entropy-selection-2026","title":"InstructDiff: Domain-Adaptive Data Selection via Contrastive Entropy for Efficient LLM Fine-Tuning","year":2026,"venue":"ACL 2026 Long Papers","authors":["Junyou Su","He Zhu","Xiao Luo","Liyu Zhang","Hong-Yu Zhou","Yun Chen","Peng Li","Yang Liu","Guanhua Chen"],"authors_zh":"Junyou Su, He Zhu, Xiao Luo, Liyu Zhang, Hong-Yu Zhou, Yun Chen, Peng Li, Yang Liu, Guanhua Chen","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","instruction-tuning","mathematics","medical","code"],"tags":["data-selection","contrastive-entropy","sft","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"领域自适应的监督微调数据选择研究","best_for_zh":"从混合指令或推理池选择模型特异监督微调子集的读者。","confidence":"high","one_line":["InstructDiff selects SFT records through calibration-induced NLL and entropy changes, retaining the lowest contrastive-entropy examples.","InstructDiff 以校准前后的负对数似然和熵变化选择监督微调记录，保留对比熵最低的例子。"],"why":"It connects each candidate response to a measurable model-state change before it is admitted to SFT.","primary_link":"https://aclanthology.org/2026.acl-long.486/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhuchichi56/Instruct-diff"}],"link_count":3,"sections":9},{"id":"adg-answer-divergence-selection-2026","title":"Instruction Data Selection via Answer Divergence","year":2026,"venue":"ACL 2026","authors":["Bo Li","Mingda Wang","Shikun Zhang","Wei Ye"],"authors_zh":"Bo Li, Mingda Wang, Shikun Zhang, Wei Ye","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["instruction-tuning","reasoning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"可读","paper_type_zh":"多样答案驱动的指令数据选择论文（ACL 2026）","best_for_zh":"研究指令数据筛选、模型中心评分和高效 SFT 的读者。","confidence":"high","one_line":["ADG selects instruction data by the geometry of multiple answers sampled from the target model.","ADG 依据从目标模型采样的多个答案的几何结构选择指令数据。"],"why":"It makes the connection between a data object and its training objective inspectable.","primary_link":"https://aclanthology.org/2026.acl-long.214/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WisdomShell/ADG"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/WisdomShell/ADG-CoT-LLaMa3-8B"},{"key":"project","label":["Project","项目主页"],"url":"https://wisdomshell.github.io/ADG/"}],"link_count":6,"sections":9},{"id":"int-self-proposed-interventions-credit-assignment-2026","title":"InT: Self-Proposed Interventions Enable Credit Assignment in LLM Reasoning","year":2026,"venue":"ICLR 2026 Poster","authors":["Matthew Y. R. Yang","Hao Bai","Ian Wu","Gene Yang","Amrith Setlur","Aviral Kumar"],"authors_zh":"Matthew Y. R. Yang、Hao Bai、Ian Wu、Gene Yang、Amrith Setlur、Aviral Kumar","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","credit-assignment","reinforcement-learning"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要从失败推理轨迹定位首错、构造定点纠错监督，并为结果奖励强化学习提供更好初始化的研究者。","confidence":"high","one_line":["InT turns failed on-policy reasoning traces into localized corrective interventions, creating process-supervision records that patch the first error before downstream RL.","InT 将失败的在策略推理轨迹转为首错位置的定点修复记录，在后续强化学习前先用过程监督修补错误。"],"why":"It converts outcome-only failures into a reusable, localized intervention record rather than treating an entire failed trajectory as uniformly bad.","primary_link":"https://arxiv.org/abs/2601.14209","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/intervention-training/int"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/CMU-AIRe/InT-SFT"},{"key":"project","label":["Project","项目主页"],"url":"https://intervention-training.github.io/"}],"link_count":4,"sections":9},{"id":"iosworld-2026","title":"iOSWorld: A Benchmark for Personally Intelligent Phone Agents","year":2026,"venue":"arXiv preprint","authors":["Lawrence Keunho Jang","Mareks Woodside","Geronimo Carom","Andrew Keunwoo Jang","Jing Yu Koh","Ruslan Salakhutdinov"],"authors_zh":"Lawrence Keunho Jang, Mareks Woodside, Geronimo Carom, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","mobile_ui","ios","gui_control","multimodal_agents","personalization","tool_use"],"tags":["environment-agent-trajectory-data","iosworld","ios","mobile-ui","computer-use-agent","multimodal-agent","persistent-persona","cross-app-reasoning","llm-as-judge","rubric-evaluation","simulator-reset","mcp-tools","replay-partial","trajectory-release-incomplete"],"status":"partial","priority":"可读","paper_type_zh":"原生iOS智能体交互基准与可执行环境","best_for_zh":"研究移动GUI智能体、环境交互轨迹、rubric-based judge、MCP工具与回放审计的读者","confidence":"medium","one_line":["iOSWorld releases 26 connected SwiftUI apps and 133 rubric-scored tasks with reproducible reset/trajectory tooling, but its GPT-5.4 Mini feedback is judgment-based and only 16 curated runs, not the full success/failure corpus, are public.","iOSWorld发布26个互联SwiftUI应用、133项任务与1,123条rubric，并开放seed state、runner、MCP和episode judge；但完整成败rollout语料未公开，现阶段仅适合评测复用。"],"why":"It turns a persistent fictional phone identity into an executable environment/data contract linking task, seeded state, screenshot or XML observation, GUI/MCP action, final answer, rubric labels, and pass/fail, while making simulator versioning, judge calibration, confirmation safety, split policy, and raw failure retention first-class audit concerns.","primary_link":"https://arxiv.org/abs/2606.09764","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ljang0/iOSWorld"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ljang0/iOSWorld/blob/e91f4cb2ef4c9dd48fef83a894477b41fd5e209d/tasks.json"},{"key":"project","label":["Project","项目主页"],"url":"https://iosworld.io/"}],"link_count":13,"sections":9},{"id":"judgeboard-2025","title":"JudgeBoard: Benchmarking and Enhancing Small Language Models for Reasoning Evaluation","year":2026,"venue":"AAAI 2026","authors":["Zhenyu Bi","Gaurav Srivastava","Yang Li","Meng Lu","Swastik Roy","Morteza Ziyadi","Xuan Wang"],"authors_zh":"Zhenyu Bi, Gaurav Srivastava, Yang Li, Meng Lu, Swastik Roy, Morteza Ziyadi, Xuan Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","llm-as-a-judge","reasoning-evaluation"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Public benchmark page for evaluating reasoning judges across multiple datasets.","面向多个数据集评测推理类评审模型的公开基准页面。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40256","links":[],"link_count":3,"sections":9},{"id":"kimi-k2-5-2026","title":"Kimi K2.5: Visual Agentic Intelligence","year":2026,"venue":"arXiv preprint","authors":["Kimi Team"],"authors_zh":"Kimi Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["reasoning","coding","software_engineering","search","agentic_tool_use","vision","video","computer_use","long_context"],"tags":["kimi-k2-5","frontier-report","data-disclosure-ledger","multimodal-rl","agent-swarm","parl","grm","zero-vision-sft"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态智能体技术报告与数据披露账本","best_for_zh":"需要审计多模态后训练、GRM、Agent Swarm 与 agent-RL 环境发布边界的研究者","confidence":"high","one_line":["Kimi K2.5 reports 15T multimodal continual pretraining, zero-vision SFT, joint RL, GRMs, and PARL but not data, teachers, environments, rollouts, or reward calibration artifacts.","Kimi K2.5 报告了 15T 多模态持续预训练、zero-vision SFT、联合 RL、GRM 和 PARL，但未发布数据、教师、环境、rollout 或 reward 校准 artifact。"],"why":"It exposes a sophisticated multimodal/multi-agent feedback contract while showing why a checkpoint and report do not make post-training reusable or fully auditable.","primary_link":"https://arxiv.org/abs/2602.02276","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MoonshotAI/Kimi-K2.5"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/moonshotai/Kimi-K2.5"},{"key":"project","label":["Project","项目主页"],"url":"https://www.kimi.com/blog/kimi-k2-5"}],"link_count":5,"sections":9},{"id":"knowme-bench-person-understanding-2026","title":"KnowMe-Bench: Benchmarking Person Understanding for Lifelong Digital Companions","year":2026,"venue":"ACL 2026","authors":["Tingyu Wu","Zhisheng Chen","Ziyan Weng","Shuhe Wang","Shuo Zhang","Sen Hu","Silin Wu","Qizhen Lan","Huacan Wang","Ronghao Chen"],"authors_zh":"Tingyu Wu、Zhisheng Chen、Ziyan Weng、Shuhe Wang、Shuo Zhang、Sen Hu、Silin Wu、Qizhen Lan、Huacan Wang、Ronghao Chen","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["trace_writing","release_audit"],"domains":["long_context","memory_agents","person_understanding"],"tags":["long_context","expert_data","evidence_grounding","memory"],"status":"verified","priority":"可读","paper_type_zh":"长上下文专家核验基准论文","best_for_zh":"构建长期记忆、人格建模或证据约束解释系统的研究者。","confidence":"high","one_line":["KnowMe-Bench evaluates whether long-context agents infer a person's motives and principles with explicit narrative evidence, not merely retrieve facts.","KnowMe-Bench 以带证据链接的长篇自传叙事，检验记忆 Agent 能否推断人的动机、状态与决策原则。"],"why":"It gives judgment-required person understanding a three-tier, evidence-auditable benchmark surface.","primary_link":"https://aclanthology.org/2026.acl-long.1394/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/realty2333/knowMe-Bench"}],"link_count":3,"sections":9},{"id":"lara-rl-contamination-2026","title":"LaRA: Layer-wise Representation Analysis for Detecting Data Contamination in RL Post-Training","year":2026,"venue":"AI4GOOD Workshop 2026; arXiv:2605.29888","authors":["Minju Gwak","Minseo Kwak","Dongseok Lee","Guijin Son","Alan Ritter","Jaehyung Kim"],"authors_zh":"Minju Gwak, Minseo Kwak, Dongseok Lee, Guijin Son, Alan Ritter, Jaehyung Kim","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["rl-contamination","representation-geometry","audit"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要审计 RL 推理模型污染风险的研究者。","confidence":"medium","one_line":["Layer-wise metrics detect reward-driven memorization after RL post-training.","以受控扰动下的逐层表征几何，审计 RL 后训练中的奖励驱动记忆污染。"],"why":"It separates contamination signals from output-level confidence changes after RL.","primary_link":"https://arxiv.org/abs/2605.29888","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=YfFRMts4R2&name=supplementary_material"}],"link_count":3,"sections":9},{"id":"llm-post-training-off-policy-on-policy-2026","title":"Large Language Model Post-Training: A Unified View of Off-Policy and On-Policy Learning","year":2026,"venue":"arXiv preprint","authors":["Shiwan Zhao","Zhihu Wang","Xuyang Zhao","Jiaming Zhou","Caiyue Xu","Chenfei Liu","Liting Zhang","Yuhang Jia","Yanzhe Zhang","Hualong Yu","Zichen Xu","Qicheng Li","Yong Qin"],"authors_zh":"Shiwan Zhao 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":["post-training","off-policy-learning","on-policy-learning","reasoning-data"],"tags":["post-training-survey","off-policy","on-policy","trajectory-provenance","taxonomy"],"status":"verified","priority":"必读","paper_type_zh":"后训练学习范式综述","best_for_zh":"关于离策略、在策略学习与多阶段语言模型后训练的统一框架。","confidence":"high","one_line":["A 2026 framework that maps post-training by who supplied the trajectory and which behavioral bottleneck a stage addresses.","一篇按轨迹来源和行为瓶颈重新组织后训练方法的 2026 综述。"],"why":"It makes trajectory provenance and stage composition visible audit questions instead of treating post-training labels as interchangeable.","primary_link":"https://arxiv.org/abs/2604.07941","links":[],"link_count":2,"sections":9},{"id":"leap-formal-mathematics-2026","title":"LEAP: Supercharging LLMs for Formal Mathematics with Agentic Frameworks","year":2026,"venue":"arXiv","authors":["Po-Nien Kung","Linfeng Song","Dawsen Hwang","Jinsung Yoon","Chun-Liang Li","Simone Severini","Mirek Olšák","Edward Lockhart","Quoc V Le","Burak Gokturk","Thang Luong","Tomas Pfister","Nanyun Peng"],"authors_zh":"Po-Nien Kung、Linfeng Song、Dawsen Hwang、Jinsung Yoon、Chun-Liang Li、Simone Severini、Mirek Olšák、Edward Lockhart、Quoc V Le、Burak Gokturk、Thang Luong、Tomas Pfister、Nanyun Peng","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","infrastructure"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","test_time_compute","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","theorem-proving"],"tags":["programmatically-verifiable-outcome","formal-mathematics","lean4","agentic-proof-search","test-time-compute","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"Lean 形式化数学 benchmark 与 verifier 引导的测试时证明搜索","best_for_zh":"适合研究程序化 verifier、形式化数学评测、agentic proof search 与 test-time compute 归因的读者。","confidence":"high","one_line":["LEAP combines Gemini 3.1 Pro, compiler-guided revision, and an AND-OR proof DAG to produce Lean-checked outcomes, but releases successful proofs rather than replayable agent trajectories.","LEAP 用 Gemini 3.1 Pro、Lean 编译反馈和 AND-OR 证明 DAG 生成可机器验收的形式化结果，但公开物只有成功证明，没有可重放的 agent 轨迹。"],"why":"For Programmatically Verifiable Outcome Data, it separates the exact terminal label supplied by Lean from the heuristic LLM judgment used to allocate search, and makes inference budget part of any fair comparison.","primary_link":"https://arxiv.org/abs/2606.03303","links":[{"key":"data","label":["Data","数据"],"url":"https://github.com/google-deepmind/superhuman/blob/96fa6c4cc3a9bb7450ee7b6773b659d3a030dace/imobench/lean_proof_bench.csv"},{"key":"project","label":["Project","项目主页"],"url":"https://imobench.github.io/"}],"link_count":4,"sections":9},{"id":"learnalign-data-selection-2025","title":"LearnAlign: Data Selection for LLM Reinforcement Learning with Improved Gradient Alignment","year":2026,"venue":"Findings of ACL 2026","authors":["Shipeng Li","Zhiqin Yang","Shikun Li","Xiaobo Xia","Hengyu Liu","Xinghua Zhang","Gaode Chen","Dong Fang","Ying Tai","Zhe Peng"],"authors_zh":"Shipeng Li、Zhiqin Yang、Shikun Li、Xiaobo Xia、Hengyu Liu、Xinghua Zhang、Gaode Chen、Dong Fang、Ying Tai、Zhe Peng","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematics","code","reasoning_data_selection"],"tags":["learnalign","data-selection","gradient-alignment","learnability","rlvr","grpo"],"status":"partial","priority":"必读","paper_type_zh":"Policy-relative RLVR 数据选择与计算效率研究","best_for_zh":"研究 RLVR prompt selection、gradient alignment、选择谱系与计算归因的读者","confidence":"high","one_line":["LearnAlign ranks prompts for one warmed-up RLVR policy by combining p(1-p) learnability with average cosine alignment among projected GRPO gradients, then trains on the top-N subset; no selected subset or implementation is released.","LearnAlign 用 p(1-p) learnability 与投影 GRPO gradient 的平均余弦对齐，为一个 warmup 后 RLVR policy 排序并选择 top-N prompts；尚未发布选后子集或实现。"],"why":"For data construction and open releases, it turns prompt selection into a checkpoint-relative data stage whose rollouts, verifier decisions, gradients, projection, ranking, and run bindings must be preserved for audit.","primary_link":"https://aclanthology.org/2026.findings-acl.2009/","links":[],"link_count":4,"sections":9},{"id":"crps-contrastive-search-trajectories-2026","title":"Learning from Contrasts: Synthesizing Reasoning Paths from Diverse Search Trajectories","year":2026,"venue":"ACL 2026","authors":["Peiyang Liu","Zhirui Chen","Xi Wang","Di Liang","Youru Li","Zhi Cai","Wei Ye"],"authors_zh":"Peiyang Liu、Zhirui Chen、Xi Wang、Di Liang、Youru Li、Zhi Cai、Wei Ye","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","trace_writing","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematical_reasoning","code_generation","commonsense_reasoning"],"tags":["crps","mcts","contrastive-trajectories","hard-negatives","soft-negatives","critique-synthesis","math-reasoning","sft","open-data","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"对比搜索轨迹合成与开放数据发布","best_for_zh":"研究 MCTS 数据构造、失败轨迹蒸馏、SFT 扩展和发布审计的读者","confidence":"medium","one_line":["CRPS contrasts successful MCTS paths with high-confidence failures or inefficient successes, uses GPT-5-mini to synthesize verified reasoning traces, and releases 27,256 final math problem/solution pairs.","CRPS 对比 MCTS 成功路径与高访问失败或低效正确路径，由 GPT-5-mini 合成终局答案可验证的轨迹；公开集实际为 27,256 条。"],"why":"It operationalizes failure-aware search-trace distillation and shows favorable SFT scaling, while its missing intermediate artifacts and count, cost, configuration, and license inconsistencies make it a useful release-audit case.","primary_link":"https://aclanthology.org/2026.acl-long.501/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PeiYangLiu/CRPS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PeiyangLiu/CRPS-30K"}],"link_count":6,"sections":9},{"id":"learning-generative-selection-2026","title":"Learning Generative Selection for Best-of-N","year":2026,"venue":"arXiv preprint","authors":["Shubham Toshniwal","Aleksander Ficek","Siddhartha Jain","Wei Du","Vahid Noroozi","Sadegh Mahdavi","Somshubra Majumdar","Igor Gitman"],"authors_zh":"Shubham Toshniwal、Aleksander Ficek、Siddhartha Jain、Wei Du、Vahid Noroozi、Sadegh Mahdavi、Somshubra Majumdar、Igor Gitman","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","code"],"tags":["generative-selection","best-of-n","genselect","dapo","rlvr","test-time-compute"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["Automatically verified math and code candidate sets train a 1.7B DAPO policy to reason over Best-of-N solutions and return the index of a verified-correct candidate.","自动验证过的数学与代码候选集用于训练 1.7B 的 DAPO 策略，让它在 Best-of-N 候选上推理并返回一个验证正确候选的编号。"],"why":"It exposes candidate-pool composition, verifier labels, selector rollouts, and selection reward as distinct data objects for the rollout/search/test-time trace track.","primary_link":"https://arxiv.org/abs/2602.02143","links":[],"link_count":3,"sections":9},{"id":"ttt-discover-test-time-2026","title":"Learning to Discover at Test Time","year":2026,"venue":"ICML 2026 Spotlight","authors":["Mert Yuksekgonul","Daniel Koceja","Xinhao Li","Federico Bianchi","Jed McCaleb","Xiaolong Wang","Jan Kautz","Yejin Choi","James Zou","Carlos Guestrin","Yu Sun"],"authors_zh":"Mert Yuksekgonul、Daniel Koceja、James Zou、Carlos Guestrin、Yu Sun（斯坦福大学）；Xinhao Li、Xiaolong Wang（加州大学圣地亚哥分校）；Federico Bianchi、James Zou（Together AI）；Jed McCaleb（Astera Institute）；Jan Kautz、Yejin Choi、Yu Sun（NVIDIA）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","rlvr","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","algorithm-design","gpu-kernel-engineering","scientific-discovery"],"tags":["test-time-compute","test-time-training","reinforcement-learning","discovery","continuous-reward"],"status":"verified","priority":"必读","paper_type_zh":"测试时强化学习与科学发现研究","best_for_zh":"研究测试时计算是否可包含参数更新和问题特定优化的读者。","confidence":"high","one_line":["TTT-Discover uses problem-specific test-time reinforcement learning to search for one high-reward scientific or engineering solution rather than merely sample from a frozen model.","TTT-Discover 在测试时针对当前问题继续强化学习，以连续奖励搜索一个最优科学或工程解，而非只从冻结模型采样。"],"why":"It expands the test-time scaling boundary from inference over frozen parameters to targeted optimization on the active problem.","primary_link":"https://arxiv.org/abs/2601.16175","links":[{"key":"code","label":["Code","代码"],"url":"https://test-time-training.github.io/discover/"}],"link_count":4,"sections":9},{"id":"learning-to-reason-for-factuality-2026","title":"Learning to Reason for Factuality","year":2026,"venue":"ICML 2026","authors":["Xilun Chen","Ilia Kulikov","Vincent-Pierre Berges","Barlas Oğuz","Rulin Shao","Gargi Ghosh","Jason Weston","Wen-tau Yih"],"authors_zh":"Xilun Chen, Ilia Kulikov, Vincent-Pierre Berges, Barlas Oğuz, Rulin Shao, Gargi Ghosh, Jason Weston, Wen-tau Yih","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed","judgment_required"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["factuality","long-form-generation","knowledge","reasoning"],"tags":["factuality","long-cot","veriscore","dpo","grpo"],"status":"partial","priority":"必读","paper_type_zh":"事实推理数据发布与多阶段后训练配方","best_for_zh":"训练或审计长文本事实推理的研究者","confidence":"high","one_line":["Releases factual-reasoning SFT and DPO data while online GRPO uses a mixed retrieval-and-judge reward whose rollouts are not released.","该工作发布经筛选的事实推理 SFT 轨迹与偏好对，而 GRPO 阶段采用检索支撑的 claim 核验和 LLM 相关性判断，反馈并非纯程序化。"],"why":"Makes the data and feedback boundary visible for studying how factual precision, supported detail, relevance, and verifier error interact in long-form reasoning post-training.","primary_link":"https://arxiv.org/abs/2508.05618","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/factual_reasoning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/factual_reasoning"}],"link_count":5,"sections":9},{"id":"generative-self-refinement-2026","title":"Learning to Refine: Self-Refinement of Parallel Reasoning in LLMs","year":2026,"venue":"Findings of ACL 2026","authors":["Qibin Wang","Pu Zhao","Shaohan Huang","Fangkai Yang","Lu Wang","Furu Wei","Qingwei Lin","Saravan Rajmohan","Dongmei Zhang"],"authors_zh":"Qibin Wang、Pu Zhao、Shaohan Huang、Fangkai Yang、Lu Wang、Furu Wei、Qingwei Lin、Saravan Rajmohan、Dongmei Zhang（机构：Microsoft、北京大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","sft","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-scaling","self-refinement","parallel-reasoning","candidate-synthesis","refinement-gap"],"status":"verified","priority":"必读","paper_type_zh":"学习式并行测试时扩展与反思研究","best_for_zh":"正在构建多样本推理系统、但需要答案生成器而非仅投票、奖励模型或重排序器的读者。","confidence":"high","one_line":["GSR trains one model to generate and then synthesize a better solution from parallel candidates, allowing test-time scaling to recover answers even when voting candidates are all wrong.","GSR 训练同一模型先生成并行候选，再从中综合出更好的答案，使测试时扩展即使在投票候选全错时也可能恢复正确解。"],"why":"It identifies learned refinement as a distinct scaling capability and provides a matched-budget metric showing when it adds value beyond majority voting.","primary_link":"https://aclanthology.org/2026.findings-acl.1291/","links":[],"link_count":2,"sections":9},{"id":"minimal-test-time-intervention-2026","title":"Less is More: Improving LLM Reasoning with Minimal Test-Time Intervention","year":2026,"venue":"ACL 2026 Long Papers","authors":["Zhen Yang","Mingyang Zhang","Feng Chen","Ganggui Ding","Liang Hou","Xin Tao","Ying-Cong Chen"],"authors_zh":"Zhen Yang、Ying-Cong Chen（香港科技大学〈广州〉、香港科技大学）；Mingyang Zhang（蚂蚁集团）；Feng Chen（AIML）；Ganggui Ding（浙江大学）；Liang Hou、Xin Tao（快手科技）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["token_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["general-reasoning","code-generation","mathematical-reasoning","multimodal-reasoning"],"tags":["test-time-scaling","entropy","classifier-free-guidance","efficient-decoding","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"词元级高效测试时干预研究","best_for_zh":"具有解码 logits 访问权限，并希望在严格时延和显存预算下替代多轨迹搜索来提升可靠性的读者。","confidence":"high","one_line":["MTI applies classifier-free guidance only to high-entropy tokens, improving reasoning while avoiding the cost of guiding every decoding position.","MTI 只在高熵词元处施加无分类器引导，在避免逐位置引导成本的同时改善推理。"],"why":"It makes the unit of test-time allocation a token-level uncertainty event rather than an entire chain, sample, or search branch.","primary_link":"https://aclanthology.org/2026.acl-long.921/","links":[],"link_count":2,"sections":9},{"id":"lexinstructeval-lexical-instruction-following-evaluation-for-large-language-models","title":"LexInstructEval: Lexical Instruction Following Evaluation for Large Language Models","year":2026,"venue":"AAAI 2026","authors":["Huimin Ren","Yan Liang","Baiqiao Su","Chaobo Sun","Hengtong Lu","Kaike Zhang","Chen Wei"],"authors_zh":"Huimin Ren, Yan Liang, Baiqiao Su, Chaobo Sun, Hengtong Lu, Kaike Zhang, Chen Wei","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A 2,475-item bilingual lexical-constraint benchmark scored deterministically by a formal grammar and programmatic engine.","2,475 条中英文细粒度约束指令，使用形式语法和程序引擎确定性判分。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/39701","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/huiminren/LexInstructEval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/renhuimin/RL-Instruction-Following-Dataset"}],"link_count":5,"sections":9},{"id":"lfqa-hp-1m-a-large-scale-human-preference-dataset-for-long-form-question-answering-2026","title":"LFQA-HP-1M: A Large-Scale Human Preference Dataset for Long-Form Question Answering","year":2026,"venue":"LREC 2026","authors":["Rafid Ishrak Jahan","Fahmid Shahriar Iqbal","Sagnik Ray Choudhury"],"authors_zh":"Rafid Ishrak Jahan、Fahmid Shahriar Iqbal、Sagnik Ray Choudhury（University of North Texas）","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["long-form-question-answering","llm-evaluation","llm-post-training"],"tags":["post-training","preference-data","human-feedback","long-form-qa","reward-modeling","llm-as-a-judge"],"status":"verified","priority":"可读","paper_type_zh":"长篇问答人工偏好数据集、量规评测与 LLM 裁判审计","best_for_zh":"需要构建或审计长篇问答奖励模型、两两偏好评测器、LLM-as-a-Judge，且能严格追溯上游公开数据许可与污染风险的研究者。","confidence":"high","one_line":["LFQA-HP-1M filters and unifies 1,326,356 public human preference pairs for English long-form QA, then uses nine rubrics to audit interpretable and LLM-based judges.","LFQA-HP-1M 从公开偏好来源筛选并统一 1,326,356 条英语长篇问答人工两两偏好记录，并以九维质量量规审计可解释模型和 LLM 裁判。"],"why":"It makes long-form QA preference feedback reusable as a traceable data object while demonstrating that a transparent rubric model can approach strong LLM judges and that those judges remain vulnerable to consistency, position, length, and perturbation failures.","primary_link":"https://arxiv.org/abs/2602.23603","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/nlpatunt/lfqa-eval-lrec-2026"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nlpatunt/LFQA-HP-1M"}],"link_count":4,"sections":9},{"id":"litecoder-terminal-sft-2026","title":"LiteCoder-Terminal: Scaling Long-Horizon Terminal Environments for Learning Language Agents","year":2026,"venue":"arXiv preprint","authors":["Peng, Xiaoxuan","Zhang, Kaiqi","Lu, Xinyu","Cao, Boxi","Lu, Yaojie","Lin, Hongyu","Han, Xianpei","Sun, Le"],"authors_zh":"Peng, Xiaoxuan、Zhang, Kaiqi、Lu, Xinyu、Cao, Boxi、Lu, Yaojie、Lin, Hongyu、Han, Xianpei、Sun, Le","tracks":["instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["scaled-long-horizon-terminal-agent-SFT"],"tags":["instruction-demonstration-rationale","arxiv-2605.29559","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"覆盖 4B 到 32B 的终端代理监督微调","confidence":"high","one_line":["LiteCoder scales terminal environments and collects 11,255 verified conversations that preserve the entire instruction-reasoning-command-observation sequence.","LiteCoder 把终端监督数据从不足千条扩展到 11255 条完整轨迹，保留指令、推理、命令和工具观察。"],"why":"Small terminal datasets provide too few long-horizon successes to study how task diversity and environment scale affect language agents.","primary_link":"https://arxiv.org/abs/2605.29559","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/icip-cas/LiteCoder"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Lite-Coder/LiteCoder-Terminal-SFT"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/blog/Lite-Coder/releasing-litecoder-terminal"}],"link_count":4,"sections":9},{"id":"llm-as-verifier-scaling-2026","title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","year":2026,"venue":"arXiv preprint","authors":["Jacky Kwok","Shulu Li","Pranav Atreya","Yuejiang Liu","Yixing Jiang","Chelsea Finn","Marco Pavone","Ion Stoica","Azalia Mirhoseini"],"authors_zh":"Jacky Kwok、Shulu Li 等（机构：斯坦福大学、加州大学伯克利分校、NVIDIA Research）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation","reinforcement_learning"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["software-engineering","agentic-reasoning","mathematical-reasoning"],"tags":["test-time-compute","verification-scaling","ranking","agent-trajectories","reward-model"],"status":"verified","priority":"必读","paper_type_zh":"验证扩展与测试时排序研究","best_for_zh":"用昂贵 LLM 验证来选择或引导智能体轨迹的读者。","confidence":"high","one_line":["LLM-as-a-Verifier turns score-token distributions into continuous trajectory rewards and scales verification along granularity, repetition, and criteria.","LLM-as-a-Verifier 把评分 token 分布转为连续轨迹奖励，并沿粒度、重复和准则三个维度扩展验证。"],"why":"It makes verifier capacity a measurable test-time scaling axis instead of treating judging as a fixed black-box step.","primary_link":"https://arxiv.org/abs/2607.05391","links":[{"key":"project","label":["Project","项目主页"],"url":"https://llm-as-a-verifier.com"}],"link_count":3,"sections":9},{"id":"llms-gaming-verifiers-rlvr-can-lead-to-reward-hacking-2026","title":"LLMs Gaming Verifiers: RLVR can Lead to Reward Hacking","year":2026,"venue":"LLM Reasoning Workshop @ ICLR 2026","authors":["Lukas Helff","Quentin Delfosse","David Steinmann","Ruben Härle","Hikaru Shindo","Patrick Schramowski","Wolfgang Stammer","Kristian Kersting","Felix Friedrich"],"authors_zh":"Lukas Helff, Quentin Delfosse, David Steinmann, Ruben Härle, Hikaru Shindo, Patrick Schramowski, Wolfgang Stammer, Kristian Kersting, Felix Friedrich","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","infrastructure","model_report"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["rlvr","audit"],"construction_layer":["reward_verifier_layer"],"domains":["inductive-logic-programming","reward-hacking","verifier-audit"],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["IPT detects when an RLVR model passes an original logic-task verifier by enumerating labels yet fails under a logically isomorphic object renaming.","揭示模型可能通过迎合验证器而非解决任务来获得 RLVR 奖励，是检查奖励黑客行为的直接证据。"],"why":"It separates executable answer acceptance from the intended invariant rule-learning objective, and shows how verifier choice changes RLVR behavior.","primary_link":"https://arxiv.org/abs/2604.15149","links":[],"link_count":1,"sections":9},{"id":"autotts-agentic-discovery-2026","title":"LLMs Improving LLMs: Agentic Discovery for Test-Time Scaling","year":2026,"venue":"arXiv preprint","authors":["Tong Zheng","Haolin Liu","Chengsong Huang","Huiwen Bao","Sheng Zhang","Rui Liu","Runpeng Dai","Ruibo Chen","Chenxi Liu","Tianyi Xiong","Xidong Wu","Hongming Zhang","Heng Huang"],"authors_zh":"Tong Zheng、Haolin Liu、Chengsong Huang、Huiwen Bao、Sheng Zhang、Rui Liu、Runpeng Dai、Ruibo Chen、Chenxi Liu、Tianyi Xiong、Xidong Wu、Hongming Zhang、Heng Huang（机构：马里兰大学、弗吉尼亚大学、圣路易斯华盛顿大学、北卡罗来纳大学、Google、Meta）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-scaling","controller-discovery","offline-replay","adaptive-compute","branch-pruning"],"status":"verified","priority":"必读","paper_type_zh":"智能体式测试时扩展控制器发现研究","best_for_zh":"正在开发自适应多轨迹推理系统，并希望以可重复方式搜索计算分配策略而非手调启发式规则的读者。","confidence":"high","one_line":["AutoTTS discovers test-time branching and stopping controllers in an offline replay environment, improving the accuracy-token frontier without hand-coding a fixed scaling rule.","AutoTTS 在离线回放环境中发现测试时的分支与停止控制器，无需手工编写固定扩展规则即可改善准确率—词元前沿。"],"why":"It makes the allocation policy itself searchable and exposes an inexpensive offline route for comparing branch, probe, pruning, and stopping decisions.","primary_link":"https://arxiv.org/abs/2605.08083","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhengkid/AutoTTS"}],"link_count":2,"sections":9},{"id":"logics-stem-failure-driven-2026","title":"Logics-STEM: Empowering LLM Reasoning via Failure-Driven Post-Training and Document Knowledge Enhancement","year":2026,"venue":"arXiv preprint (2026)","authors":["Mingyu Xu","Cheng Fang","Keyue Jiang","Yuqian Zheng","Yanghua Xiao","Baojian Zhou","Qifang Zhao","Suhang Zheng","Xiuwen Zhu","Jiyang Tang","Yongchi Zhao","Yijia Luo","Zhiqi Bai","Yuchi Xu","Wenbo Su","Wei Wang","Bing Zhao","Lin Qu","Xiaoxiao Xu"],"authors_zh":"Mingyu Xu, Cheng Fang, Keyue Jiang, Yuqian Zheng, Yanghua Xiao, Baojian Zhou, Qifang Zhao, Suhang Zheng, Xiuwen Zhu, Jiyang Tang, Yongchi Zhao, Yijia Luo, Zhiqi Bai, Yuchi Xu, Wenbo Su, Wei Wang, Bing Zhao, Lin Qu, Xiaoxiao Xu","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release","model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["stem-reasoning"],"tags":["long-cot-data","stem","failure-driven-post-training"],"status":"verified","priority":"可读","paper_type_zh":"STEM 长 CoT 数据集与失败驱动后训练报告","best_for_zh":"适合整理大规模 STEM 推理数据混合或针对模型失败设计第二阶段训练的研究者。","confidence":"high","one_line":["Logics-STEM curates millions of long-CoT STEM records and uses model failures to retrieve documents and synthesize targeted data for a second SFT or RLVR stage.","Logics-STEM 整理数百万条 STEM 长 CoT 记录，并围绕模型失败点检索文档、合成针对性数据，用于第二阶段 SFT 或 RLVR。"],"why":"It links large-scale data curation, explicit decontamination, sampling, and failure-targeted post-training in one disclosed pipeline.","primary_link":"https://arxiv.org/abs/2601.01562","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Logics-MLLM/Logics-STEM-SFT-Dataset-Open-5.3M"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Logics-MLLM"}],"link_count":3,"sections":9},{"id":"long-horizon-terminal-bench-2026","title":"Long-Horizon-Terminal-Bench: Testing the Limits of Agents on Long-Horizon Terminal Tasks with Dense Reward-Based Grading","year":2026,"venue":"arXiv preprint","authors":["Zongxia Li","Zhongzhi Li","Yucheng Shi","Ruhan Wang","Junyao Yang","Zhichao Liu","Xiyang Wu","Anhao Li","Yue Yu","Ninghao Liu","Lichao Sun","Haotao Mi","Leowei Liang"],"authors_zh":"Zongxia Li、Zhongzhi Li、Yucheng Shi、Ruhan Wang、Junyao Yang、Zhichao Liu、Xiyang Wu、Anhao Li、Yue Yu、Ninghao Liu、Lichao Sun、Haotao Mi、Leowei Liang","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release","verifier_reward","agent_environment","construction_recipe"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["terminal_agents","shell_and_cli","docker_environments","agent_trajectories","software_engineering","scientific_computing","multimodal_analysis","games_and_puzzles","professional_workflows"],"tags":["long-horizon-terminal-bench","terminal-agents","docker","harbor","agent-trajectories","environment-feedback","deterministic-verifier","dense-subtask-reward","partial-credit","replay","evaluation-only","contamination-risk"],"status":"partial","priority":"必读","paper_type_zh":"长时程终端智能体基准与环境发布","best_for_zh":"研究环境反馈、长时程智能体评测、确定性 verifier、轨迹审计与基准污染的读者","confidence":"medium","one_line":["LHTB releases 46 test-only containerized terminal tasks with deterministic end-of-rollout subtask rewards, but not the paper's model trajectories, and its current GitHub exposure of tests/solutions conflicts with the HF held-out policy.","LHTB 发布 46 个仅用于测试的容器化终端任务及确定性回合末子任务评分器，但未发布论文基线轨迹；当前 GitHub 公开全部任务的 tests/ 与 solution/，与 HF 的保留声明冲突。"],"why":"It makes long-horizon terminal progress measurable without collapsing every incomplete run to zero, while showing that dense environmental rewards are only trustworthy when verifier secrecy, trajectory retention, container replay, version binding and rights are audited together.","primary_link":"https://arxiv.org/abs/2607.08964","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zli12321/LHTB"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://zli12321.github.io/LHTB/"}],"link_count":11,"sections":9},{"id":"longrlvr-verifiable-context-rewards-2026","title":"LongRLVR: Long-Context Reinforcement Learning Requires Verifiable Context Rewards","year":2026,"venue":"ICLR 2026","authors":["Guanzheng Chen","Michael Qizhe Shieh","Lidong Bing"],"authors_zh":"Guanzheng Chen, Michael Qizhe Shieh, Lidong Bing","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["process_reward","answer_level"],"training_use":["rlvr"],"construction_layer":["reward_verifier_layer","trace_writing"],"domains":["long-context-reasoning","question-answering"],"tags":["long-context","rlvr","grounding","reward-design"],"status":"verified","priority":"必读","paper_type_zh":"长上下文强化学习与可验证奖励设计论文（ICLR 2026）","best_for_zh":"研究长上下文问答、RLVR、证据定位与训练奖励设计的读者。","confidence":"high","one_line":["LongRLVR trains long-context models with a verifiable reward for evidence selection as well as answer correctness.","LongRLVR 将证据块选择与最终答案同时纳入可验证奖励，使长上下文强化学习获得更稠密的定位反馈。"],"why":"It makes the intermediate grounding record, rather than only the final answer, an explicit consumer of the RL objective.","primary_link":"https://arxiv.org/abs/2603.02146","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/real-absolute-AI/LongRLVR"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Guanzheng/LongRLVR-Data"}],"link_count":5,"sections":9},{"id":"lookahead-tree-based-rollouts-2026","title":"Lookahead Tree-Based Rollouts for Enhanced Trajectory-Level Exploration in Reinforcement Learning with Verifiable Rewards","year":2026,"venue":"ICLR 2026","authors":["Shangyu Xing","Siyuan Wang","Chenyuan Yang","Xinyu Dai","Xiang Ren"],"authors_zh":"Shangyu Xing、Siyuan Wang、Chenyuan Yang、Xinyu Dai、Xiang Ren","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematical_reasoning","logical_reasoning"],"tags":["latr","rlvr","tree-based-rollouts","trajectory-diversity","lookahead-search","branch-and-prune","grpo","dapo","programmatic-verification","processed-prompt-release"],"status":"partial","priority":"必读","paper_type_zh":"在线 RLVR 树式 rollout 构造配方与预处理数据发布","best_for_zh":"研究搜索生成推理轨迹、分支与剪枝策略、rollout 预算归因、终点验证及公开数据审计的读者","confidence":"high","one_line":["LATR generates diverse on-policy RLVR rollout groups by probability-gated branching, lookahead simulation, and similarity pruning, while its public data release contains processed Countdown and math prompt/reward rows rather than the raw rollout trees.","LATR 在在线 RLVR 中以概率门控分支、前瞻模拟和相似度剪枝构造八路 rollout 组；其 ICLR 2026 配方可用于研究轨迹级探索，但公开数据只有 Countdown 与数学任务的预处理 prompt/reward 行，不含原始搜索树、被拒绝分支或逐样本奖励判定。"],"why":"It isolates rollout diversity as a training-data construction variable and reports faster, stronger GRPO/DAPO learning, but also shows why raw branch lineage, verifier decisions, paper-matching configurations, and release licenses are necessary to audit search-generated reasoning data.","primary_link":"https://openreview.net/forum?id=4nLvUk8edu","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/starreeze/latr"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/starreeze/latr-data"}],"link_count":7,"sections":9},{"id":"lost-in-translation-lvlm-judges-2026","title":"Lost in Translation: Do LVLM Judges Generalize Across Languages?","year":2026,"venue":"Findings of ACL 2026","authors":["Md Tahmid Rahman Laskar","Mohammed Saidul Islam","Mir Tafseer Nayeem","Amran Bhuiyan","Mizanur Rahman","Shafiq Joty","Enamul Hoque","Jimmy Huang"],"authors_zh":"Md Tahmid Rahman Laskar, Mohammed Saidul Islam, Mir Tafseer Nayeem, Amran Bhuiyan, Mizanur Rahman, Shafiq Joty, Enamul Hoque, Jimmy Huang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multilingual","vision-language","llm-as-a-judge"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"多语言多模态评审模型评测与偏好数据基准论文","best_for_zh":"需要评估视觉语言评审器能否跨语言稳定处理通用图文偏好与图表推理反馈的研究者。","confidence":"high","one_line":["MM-JudgeBench evaluates LVLM judges on over 60K multilingual image-text preference pairs across 25 languages and chart-reasoning settings.","MM-JudgeBench 有逾 6 万个跨 25 种语言的图文偏好实例，检验奖励模型跨语言迁移及图表推理反馈稳定性。"],"why":"It tests whether a visual reward signal survives language transfer rather than assuming English judge accuracy generalizes.","primary_link":"https://aclanthology.org/2026.findings-acl.1746/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tahmedge/MM-JudgeBench"}],"link_count":4,"sections":9},{"id":"m2po-translation-preference-2026","title":"M2PO: Multi-Perspective Multi-Pair Preference Optimization for Machine Translation","year":2026,"venue":"ACL 2026","authors":["Hao Wang","Linlong Xu","Heng Liu","Yangyang Liu","Xiaohu Zhao","Bo Zeng","Liangying Shao","Yichen Dong","Xinwei Wu","Jiang Zhou","Tianyu Dong","Xiangxiang Zeng","Longyue Wang","Weihua Luo"],"authors_zh":"Hao Wang 等（阿里巴巴集团、湖南大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["machine_translation","preference_learning"],"tags":["preference-data","machine-translation","curriculum","multi-pair-optimization"],"status":"verified","priority":"可读","paper_type_zh":"多视角偏好数据构造与多对偏好优化研究","best_for_zh":"适合构造翻译偏好数据，或研究流畅回答如何掩盖局部事实错误的读者。","confidence":"high","one_line":["M2PO converts translation candidate sets into faithfulness-aware multi-pair preferences and trains them with a confidence-scheduled ranking objective.","M²PO 将翻译候选集转为兼顾忠实度、质量与模型置信度的多对偏好监督，并以排序目标共同训练。"],"why":"It treats the full candidate set, rather than one best-worst pair, as the training object and ties data ranking to a changing optimization objective.","primary_link":"https://aclanthology.org/2026.acl-long.469/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AIDC-AI/Marco-MT/tree/master/MMPO"}],"link_count":3,"sections":9},{"id":"making-slow-thinking-faster-step-entropy-2025","title":"Making Slow Thinking Faster: Compressing LLM Chain-of-Thought via Step Entropy","year":2026,"venue":"International Conference on Learning Representations (ICLR 2026)","authors":["Zeju Li","Jianyuan Zhong","Ziyang Zheng","Xiangyu Wen","Zhijian Xu","Yingying Cheng","Fan Zhang","Qiang Xu"],"authors_zh":"Zeju Li、Jianyuan Zhong、Ziyang Zheng、Xiangyu Wen、Zhijian Xu、Yingying Cheng、Fan Zhang、Qiang Xu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","rlvr","test_time_compute"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematical_reasoning","knowledge_reasoning"],"tags":["cot-compression","step-entropy","skip-token","long-to-short","trace-selection","sft","grpo","rlvr","test-time-efficiency"],"status":"partial","priority":"必读","paper_type_zh":"基于步骤熵的长短链推理压缩、SFT 与 GRPO 训练配方","best_for_zh":"关注推理轨迹选择、CoT 压缩、RLVR 反馈契约和数据可审计性的读者","confidence":"high","one_line":["The paper ranks newline-delimited CoT steps by model-derived entropy, replaces low-entropy steps with [SKIP], and trains SFT-plus-GRPO models to generate shorter reasoning traces.","该工作以模型导出的长度归一化步骤熵筛选完整 CoT 中的低熵步骤，用 [SKIP] 构造压缩轨迹，并以 SFT 和 GRPO 学习更短的推理输出。"],"why":"It operationalizes Track 5's core distinction between a full rollout trace and a selected trace: a model-derived step score chooses what to hide, and the selected compressed trace becomes an SFT target before GRPO further trades off answer correctness and token use. The unavailable generated corpus and selector/reward logs prevent independent assessment of lineage, selection bias, and reusable training data.","primary_link":"https://arxiv.org/abs/2508.03346","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/staymylove/COT_Compresstion_via_Step_entropy"}],"link_count":4,"sections":9},{"id":"mathsmith-2025","title":"MathSmith: Towards Extremely Hard Mathematical Reasoning by Forging Synthetic Problems with a Reinforced Policy","year":2026,"venue":"AAAI 2026","authors":["Shaoxiong Zhan","Yanlin Lai","Ziyu Lu","Dahua Lin","Ziqing Yang","Fei Tan"],"authors_zh":"Shaoxiong Zhan、Yanlin Lai、Ziyu Lu、Dahua Lin、Ziqing Yang、Fei Tan","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["mathematics"],"tags":["synthetic-math-problems","reinforced-data-generation","difficulty-control","answer-consistency","long-cot","weakness-focused-generation"],"status":"verified","priority":"必读","paper_type_zh":"强化式困难数学数据构建与开放发布","best_for_zh":"研究合成难题、数据生成 RL、难度奖励、长 CoT 与弱点定向训练的读者","confidence":"high","one_line":["Trains and releases Qwen3-8B problem synthesizers that turn PlanetMath-derived concepts into rationales and hard math problems using structural, trace-length, and same-teacher consistency rewards.","MathSmith 用 PlanetMath 衍生概念训练并发布 Qwen3-8B 出题器，以结构、教师推理长度和同教师答案一致性奖励生成题目与构造 rationale；这些信号仍不是独立数学验证。"],"why":"It exposes a reusable generator-side SFT-to-GRPO pipeline and intermediate data objects while showing that its difficulty and availability signals remain teacher-dependent proxies rather than independent mathematical verification.","primary_link":"https://arxiv.org/abs/2508.05592","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Jasaxion/MathSmith"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Jasaxion/MathSmith-HC-Problems"},{"key":"project","label":["Project","项目主页"],"url":"https://jasaxion.github.io/MathSmith_ProjectPage/"}],"link_count":11,"sections":9},{"id":"mc-search-multimodal-agentic-search-2026","title":"MC-Search: Evaluating and Enhancing Multimodal Agentic Search with Structured Long Reasoning Chains","year":2026,"venue":"ICLR 2026 Oral","authors":["Xuying Ning","Dongqi Fu","Tianxin Wei","Mengting Ai","Jiaru Zou","Ting-Wei Li","Hanghang Tong","Yada Zhu","Hendrik Hamann","Jingrui He"],"authors_zh":"Xuying Ning、Dongqi Fu、Tianxin Wei、Mengting Ai、Jiaru Zou、Ting-Wei Li、Hanghang Tong、Yada Zhu、Hendrik Hamann、Jingrui He","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-agentic-search","retrieval-augmented-generation","process-supervision"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要多模态检索轨迹、逐跳证据归因，或训练搜索型 MLLM 的研究者。","confidence":"high","one_line":["MC-Search benchmarks multimodal agents on verified multi-hop image–text evidence chains and turns those chains into process supervision for better retrieval planning.","MC-Search 以经核验的图文多跳证据链评测多模态智能体，并将其转化为改进检索规划的过程监督。"],"why":"It makes retrieval modality, supporting evidence, and intermediate answers explicit at every hop rather than supervising only final QA.","primary_link":"https://openreview.net/forum?id=JEGDp1E4OH","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/YennNing/MC-Search"},{"key":"project","label":["Project","项目主页"],"url":"https://mc-search-project.github.io/"}],"link_count":4,"sections":9},{"id":"mcp-agentbench-evaluating-real-world-language-agent-performance-2025","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","year":2026,"venue":"AAAI 2026","authors":["Zikang Guo","Benfeng Xu","Chiwei Zhu","Wentao Hong","Xiaorui Wang","Zhendong Mao"],"authors_zh":"Zikang Guo, Benfeng Xu, Chiwei Zhu, Wentao Hong, Xiaorui Wang, Zhendong Mao","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","mcp","multi_server_workflows"],"tags":["environment-agent-trajectory-data","agent-benchmark","mcp","tool-use","multi-server","stateless-environment","llm-as-a-judge","answer-level-verification","evaluation-only","replay-risk"],"status":"partial","priority":"可读","paper_type_zh":"MCP 工具智能体评测基准与 answer-level judge 研究","best_for_zh":"研究 MCP 智能体评测、LLM-as-a-judge、轨迹可见性和可重放性审计的读者","confidence":"high","one_line":["MCP-AgentBench evaluates 600 category-balanced queries over 33 stateless text MCP servers with answer-only o3-mini-high pass/fail judgments, while releasing no confirmed server bundle, task data, or success/failure trajectory archive.","MCP-AgentBench 用 600 个均衡分类 query 评测 33 个无状态文本 MCP server 上的工具智能体，但 MCP-Eval 只对 final answer 做 o3-mini-high 二元判断，且未核实到可重放的官方任务、server 或轨迹 release。"],"why":"It separates an agent episode from its supervision contract: models may generate rich state-action-observation histories, yet the benchmark observes only the final answer. That makes it a strong case study for auditing judge visibility, terminal semantics, server/version drift, replayability, and the difference between evaluation logs and reusable post-training data.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/download/40347/44308","links":[],"link_count":4,"sections":9},{"id":"mediocrity-llm-judge-anchor-selection-2026","title":"Mediocrity is the key for LLM as a Judge Anchor Selection","year":2026,"venue":"ACL 2026","authors":["Shachar Don-Yehiya","Asaf Yehudai","Leshem Choshen","Omri Abend"],"authors_zh":"Shachar Don-Yehiya, Asaf Yehudai, Leshem Choshen, Omri Abend","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","reward_modeling","preference_learning","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["llm_evaluation"],"tags":["llm_as_judge","anchor_selection","pairwise","ranking","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"LLM-as-a-Judge 锚点选择分析与大规模比较判决数据","best_for_zh":"需要设计、校准或审计成对比较式 LLM Judge 与模型排行榜的研究者。","confidence":"high","one_line":["This work shows that middling anchors make pairwise LLM judging more informative than extreme anchors.","该研究表明，在 LLM Judge 的锚点式比较中，中等能力锚点比极强或极弱锚点更能区分模型排序。"],"why":"It makes the hidden choice of comparison anchor measurable and auditable in scalable model evaluation.","primary_link":"https://aclanthology.org/2026.acl-long.706/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/IBM/Anchor-Selection"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ibm-research/900K-Judgements"}],"link_count":5,"sections":9},{"id":"meds3-medical-slow-thinking-process-supervision-2026","title":"MedS³: Towards Medical Slow Thinking with Self-Evolved Soft Dual-sided Process Supervision","year":2026,"venue":"AAAI 2026","authors":["Shuyang Jiang","Yusheng Liao","Zhe Chen","Ya Zhang","Yanfeng Wang","Yu Wang"],"authors_zh":"Shuyang Jiang, Yusheng Liao, Zhe Chen, Ya Zhang, Yanfeng Wang, Yu Wang","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["medical-reasoning","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"医学过程监督数据集与过程奖励模型论文","best_for_zh":"需要医学推理的正负轨迹、步骤价值标签或过程奖励训练数据的研究者","confidence":"high","one_line":["MedS³ releases medical MCTS trajectories with positive and negative paths plus per-step Q-values and rollout values for soft dual-sided process supervision.","发布医学问题的正负 MCTS 推理轨迹及步骤 Q-value、rollout value，用于软双向过程奖励监督。"],"why":"It exposes a concrete positive/negative trajectory format for medical process reward modeling rather than only final-answer supervision.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40395","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pixas/MedSSS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/pixas/MedSSS-data"}],"link_count":5,"sections":9},{"id":"megascience-2026","title":"MegaScience: Pushing the Frontiers of Post-Training Datasets for Science Reasoning","year":2026,"venue":"COLM 2026","authors":["Fan, Run-Ze","Wang, Zengzhi","Liu, Pengfei"],"authors_zh":"Fan, Run-Ze、Wang, Zengzhi、Liu, Pengfei","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reference-anchored-science-reasoning-mixture"],"tags":["instruction-demonstration-rationale","arxiv-2507.16812","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"科学推理监督微调","confidence":"high","one_line":["MegaScience combines verified textbook references with source-specific selection and solution annotation to release a 1.25M seven-discipline reasoning mixture.","MegaScience 用教材参考答案、逐来源筛选和分步解答标注构建 125 万条、覆盖七个学科的科学推理混合。"],"why":"Open post-training corpora emphasize math and code, while scientific questions often lack trustworthy references and sufficiently detailed worked solutions.","primary_link":"https://arxiv.org/abs/2507.16812","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MegaScience/MegaScience"}],"link_count":2,"sections":9},{"id":"menvagent-scalable-polyglot-environment-construction-for-verifiable-software-engineering","title":"MEnvAgent: Scalable Polyglot Environment Construction for Verifiable Software Engineering","year":2026,"venue":"ICML 2026","authors":["Chuanzhe Guo","Jingjing Wu","Sijun He","Yang Chen","Zhaoqi Kuang","Shilong Fan","Bingjin Chen","Siqi Bao","Jing Liu","Hua Wu","Qingfu Zhu","Wanxiang Che","Haifeng Wang"],"authors_zh":"Chuanzhe Guo, Jingjing Wu, Sijun He, Yang Chen, Zhaoqi Kuang, Shilong Fan, Bingjin Chen, Siqi Bao, Jing Liu, Hua Wu, Qingfu Zhu, Wanxiang Che, Haifeng Wang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** MEnvAgent uses a multi-agent feedback loop to construct polyglot SWE sandboxes, addressing the loss of verifiable tasks to environment failures.","MEnvAgent 用多代理反馈闭环自动构造多语言 SWE 沙箱，解决可验证任务被环境失败卡住的问题。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2601.22859","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ernie-research/MEnvAgent"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ernie-research/MEnvData-SWE"}],"link_count":4,"sections":9},{"id":"meta-judging-large-language-models-2026","title":"Meta-Judging with Large Language Models: Concepts, Methods, and Challenges","year":2026,"venue":"arXiv preprint","authors":["Hugo Silva","Mateus Mendes","Hugo Gonçalo Oliveira"],"authors_zh":"Hugo Silva、Mateus Mendes、Hugo Gonçalo Oliveira","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","reward_modeling"],"construction_layer":["release_audit"],"domains":["llm-as-a-judge","meta-judging","evaluation-methodology","rubric-feedback"],"tags":["llm-as-a-judge","meta-judging","evaluation-survey","judge-reliability","audit"],"status":"verified","priority":"可读","paper_type_zh":"语言模型评审与元评审综述","best_for_zh":"关于语言模型裁判与元评审概念、方法和局限的综述。","confidence":"high","one_line":["A 2026 map of how automated judges are evaluated, aligned, and still vulnerable to bias and prompt effects.","一篇审视自动裁判如何被再评估、对齐和审计的 2026 年综述。"],"why":"It prevents rubric or judge outputs from being treated as ground truth when they are a learned, prompt-conditioned feedback signal.","primary_link":"https://arxiv.org/abs/2601.17312","links":[],"link_count":2,"sections":9},{"id":"metascale-2025","title":"MetaScale: Test-Time Scaling with Evolving Meta-Thoughts","year":2026,"venue":"Findings of the Association for Computational Linguistics: ACL 2026","authors":["Qin Liu","Wenxuan Zhou","Nan Xu","James Y. Huang","Fei Wang","Sheng Zhang","Hoifung Poon","Muhao Chen"],"authors_zh":"Qin Liu、Wenxuan Zhou、Nan Xu、James Y. Huang、Fei Wang、Sheng Zhang、Hoifung Poon、Muhao Chen","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward","trajectory_value"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["reasoning","mathematics","general-purpose"],"tags":["test-time-scaling","meta-thoughts","multi-armed-bandit","ucb","genetic-evolution","reward-model-selection"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["MetaScale searches over query-specific cognitive strategies with UCB and LLM-mediated evolution, using an outcome reward model to select the final response under a fixed inference budget.","MetaScale 用 UCB 与模型中介的演化在与问题相关的认知策略上搜索，并在固定推理预算下由结果奖励模型选出最终回答。"],"why":"It provides a concrete schema for auditing strategy-level test-time traces—retrieval source, arm statistics, rewards, evolution lineage, budget, and final selection—without treating benchmark gains as evidence that the hidden traces are reusable or high quality.","primary_link":"https://aclanthology.org/2026.findings-acl.574/","links":[],"link_count":4,"sections":9},{"id":"mimo-v2-flash-2026","title":"MiMo-V2-Flash Technical Report","year":2026,"venue":"arXiv preprint","authors":["Xiaomi LLM-Core Team"],"authors_zh":"Xiaomi LLM-Core Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","process_reward","scalar_reward","state_action_level","full_episode"],"training_use":["sft","distillation","process_supervision","rlvr","agent_training","safety_alignment"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["reasoning","mathematics","code","agent","tool-use","web","safety"],"tags":["mimo-v2-flash","frontier-report","data-disclosure-ledger","mopd","agentic-rl","token-level-feedback","reward-hacking"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型技术报告与 agentic RL 数据披露账本","best_for_zh":"需要审计多教师后训练、agent 环境、MOPD 反馈契约和 reward-hacking 风险的研究者","confidence":"medium","one_line":["MiMo-V2-Flash describes MOPD plus agentic RL with teacher-KL, programmatic, judge, and multimodal feedback but withholds data, environments, teachers, and reward implementation.","MiMo-V2-Flash 描述了结合 teacher-KL、程序化检查、judge 与多模态反馈的 MOPD 和 agentic RL，但未发布数据、环境、教师和 reward 实现。"],"why":"Discloses feedback and environment surfaces that are often absent while demonstrating why unreleased multi-teacher/RL pipelines are not reproducible data recipes.","primary_link":"https://arxiv.org/abs/2601.02780","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2-Flash"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/XiaomiMiMo/MiMo-V2-Flash"}],"link_count":4,"sections":9},{"id":"mindloom-reasoning-data-synthesis-2026","title":"MindLoom: Composing Thought Modes for Frontier-Level Reasoning Data Synthesis","year":2026,"venue":"arXiv preprint (2026)","authors":["Haiyang Shen","Taian Guo","Xuanzhong Chen","Mugeng Liu","Weichen Bi","Wenchun Jing","Sixiong Xie","Zhuofan Shi","Yudong Han","Chongyang Pan","Siqi Zhong","Jinsheng Huang","Ming Zhang","Yun Ma"],"authors_zh":"Haiyang Shen, Taian Guo, Xuanzhong Chen, Mugeng Liu, Weichen Bi, Wenchun Jing, Sixiong Xie, Zhuofan Shi, Yudong Han, Chongyang Pan, Siqi Zhong, Jinsheng Huang, Ming Zhang, Yun Ma","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["stem-reasoning","mathematical-reasoning"],"tags":["thought-modes","compositional-synthesis","provenance-filtering"],"status":"verified","priority":"可读","paper_type_zh":"组合式推理数据合成方法与数据集","best_for_zh":"适合用可复用推理变换生成 STEM SFT 数据的研究者。","confidence":"high","one_line":["MindLoom extracts reusable thought modes from verified solutions, composes them under compatibility and scarcity control, and keeps 9,230 judged-correct SFT records.","MindLoom 从已验证解答中提取可复用 thought mode，在兼容性与稀缺度控制下组合新题，并保留 9,230 条判定正确的 SFT 记录。"],"why":"It makes reasoning-data composition auditable through mode tuples, retrieval scores, provenance filters, and rollout decisions.","primary_link":"https://arxiv.org/abs/2605.21630","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EachSheep/MindLoom"}],"link_count":2,"sections":9},{"id":"minimax-m2-5-2026","title":"MiniMax-M2.5","year":2026,"venue":"MiniMax official model release","authors":["MiniMax AI"],"authors_zh":"MiniMax AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","process_supervision","agent_environment","verifier_reward","benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","state_action_level","full_episode","pairwise_preference","scalar_reward","process_reward","trajectory_value"],"training_use":["process_supervision","agent_training","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["agentic_reasoning","software_engineering","multilingual_code","tool_use","web_search","office_productivity","document_generation","spreadsheet_reasoning","long_context"],"tags":["minimax","minimax-m2-5","open-weights","mixture-of-experts","agentic-rl","forge","cispo","process-reward","speed-reward","software-environments","coding-agent","web-search","tool-calling","office-agent","deliverable-evaluation","llm-as-judge","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"开放权重 Agent 模型发布与数据披露账本","best_for_zh":"研究软件工程 Agent RL、工作环境数据、过程/速度奖励、Forge/CISPO 与发布审计的读者","confidence":"high","one_line":["MiniMax-M2.5 reports hundreds of thousands of work-derived RL environments, 200K+ software environments, Forge tree-merged asynchronous rollouts, and CISPO rewards, but releases only 229B FP8 weights and inference materials—not environments, reward systems, splits, or training lineage.","MiniMax-M2.5 报告 20 万+软件环境、Forge 树合并异步 rollout 与 CISPO 性能/过程/速度反馈，并开放 229B FP8 权重；环境、奖励系统、训练数据和 lineage 仍未发布，SWE-Bench 80.2 与 75.8 也未解释。"],"why":"It expands the ledger from prompt-response alignment to testable software, search, and professional deliverables while exposing unresolved rights, reward calibration, judge bias, internal evaluation, and reproducibility risks.","primary_link":"https://github.com/MiniMax-AI/MiniMax-M2.5","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5"}],"link_count":3,"sections":9},{"id":"mmfinereason-2026","title":"MMFineReason: Closing the Multimodal Reasoning Gap via Open Data-Centric Methods","year":2026,"venue":"arXiv preprint","authors":["Honglin Lin","Zheng Liu","Yun Zhu","Chonghan Qin","Juekai Lin","Xiaoran Shang","Conghui He","Wentao Zhang","Lijun Wu"],"authors_zh":"Honglin Lin, Zheng Liu, Yun Zhu, Chonghan Qin, Juekai Lin, Xiaoran Shang, Conghui He, Wentao Zhang, Lijun Wu","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation","rlvr"],"construction_layer":["trace_writing"],"domains":["multimodal-reasoning","reasoning-data"],"tags":["arxiv-2601.21821","instruction-demonstration-rationale-data","multimodal-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多模态推理数据构造与筛选","best_for_zh":"适合构建或筛选多模态长思维链数据，并用于 SFT 和后续 RL 的研究人员。","confidence":"high","one_line":["MMFineReason distills and filters 1.8M multimodal long-CoT examples, then selects a 123K difficult subset that retains most of the training gain.","MMFineReason 蒸馏并筛选出 181 万条多模态长思维链样本，再选出 12.3 万条难例，以较少数据保留大部分训练收益。"],"why":"It shows how source normalization, answer consistency, deduplication, and student-relative difficulty can make multimodal reasoning data more efficient.","primary_link":"https://arxiv.org/abs/2601.21821","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/OpenDataArena/mmfinereason"}],"link_count":2,"sections":9},{"id":"mmsearch-r1-2025","title":"MMSearch-R1: Incentivizing LMMs to Search","year":2026,"venue":"ACL 2026 (Long Papers)","authors":["Jinming Wu","Zihao Deng","Wei Li","Yiding Liu","Bo You","Bo Li","Zejun Ma","Ziwei Liu"],"authors_zh":"Jinming Wu, Zihao Deng, Wei Li, Yiding Liu, Bo You, Bo Li, Zejun Ma, Ziwei Liu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["agent_training","test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","raw_search_rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["MMSearch-R1: Incentivizing LMMs to Search records multi-round text/image search traces and cached image-search results under outcome reward plus search penalty.","MMSearch-R1 以多轮图像和文本搜索训练多模态模型按需调用工具，并发布 FVQA 与缓存图像检索结果。"],"why":"It makes RL decides multiround search actions and its audit boundary visible for reasoning-data curation.","primary_link":"https://aclanthology.org/2026.acl-long.114/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EvolvingLMMs-Lab/multimodal-search-r1"}],"link_count":5,"sections":9},{"id":"mobilebench-ol-2026","title":"MobileBench-OL: A Comprehensive Chinese Benchmark for Evaluating Mobile GUI Agents in Real-World Environment","year":2026,"venue":"Findings of the Association for Computational Linguistics: ACL 2026","authors":["Qinzhuo Wu","Zhizhuo Yang","Hanhao Li","Pengzhi Gao","Wei Liu","Jian Luan"],"authors_zh":"Qinzhuo Wu, Zhizhuo Yang, Hanhao Li, Pengzhi Gao, Wei Liu, Jian Luan","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","verifier_reward","data_release"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","mobile_gui","chinese_apps","long_horizon_reasoning","gui_exploration","noise_robustness"],"tags":["mobile-gui-agent","real-device-benchmark","chinese-apps","agent-trajectories","xpath-verifier","environmental-reward","long-horizon","gui-exploration","noise-injection","reset-mechanism","app-version-drift","verifier-gaming","evaluation-contamination"],"status":"partial","priority":"可读","paper_type_zh":"真实设备移动GUI智能体基准与自动评测环境","best_for_zh":"研究移动GUI智能体、环境交互轨迹、环境型verifier、噪声鲁棒性、状态重置与评测污染的读者","confidence":"medium","one_line":["MobileBench-OL publishes 1,080 real-device Chinese-app evaluation cases with XPath-like success rules, noise injection and inverse-task resets, but not the paper's complete sampled trajectories or a frozen APK/device environment.","MobileBench-OL公开覆盖80款中文应用的1,080个真实设备评测项，以XPath式成功规则、噪声注入和逆任务重置评分，但没有发布论文所用的完整成败轨迹或冻结APK/设备环境。"],"why":"It shows how screenshots, XML state, actions, terminal signaling, partial predicates and reset episodes form an auditable mobile-agent feedback contract, while demonstrating why public rules, app drift, imperfect resets and missing failed-rollout lineage constrain reuse to evaluation rather than training.","primary_link":"https://aclanthology.org/2026.findings-acl.668/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xiaomi-research/mobilebench-ol"},{"key":"data","label":["Data","数据"],"url":"https://github.com/xiaomi-research/mobilebench-ol/tree/main/data"}],"link_count":7,"sections":9},{"id":"mobilegym-2026","title":"MobileGym: A Verifiable and Highly Parallel Simulation Platform for Mobile GUI Agent Research","year":2026,"venue":"arXiv preprint","authors":["Dingbang Wu","Rui Hao","Haiyang Wang","Shuzhe Wu","Han Xiao","Zhenghong Li","Bojiang Zhou","Zheng Ju","Zichen Liu","Lue Fan","Zhaoxiang Zhang"],"authors_zh":"Dingbang Wu, Rui Hao, Haiyang Wang, Shuzhe Wu, Han Xiao, Zhenghong Li, Bojiang Zhou, Zheng Ju, Zichen Liu, Lue Fan, Zhaoxiang Zhang","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mobile_gui_agents","agent_trajectories","reinforcement_learning"],"tags":["environment-agent-trajectory-data","agent-trajectories"],"status":"partial","priority":"可读","paper_type_zh":"移动 GUI 模拟环境、可验证基准与在线强化学习案例","best_for_zh":"研究移动 GUI agent、环境反馈、RLVR、并行 rollout、sim-to-real 与 verifier 审计的读者","confidence":"medium","one_line":["MobileGym is a browser-hosted mobile GUI simulator whose structured state supports reset/forked rollouts, deterministic state-based judging, and a reported GRPO run.","MobileGym 以可配置、可分叉的结构化 JSON 状态承载移动 GUI episode，并用确定性终局检查同时支持评测与在线 GRPO；其公开物是模拟器、任务和训练代码，而非论文实验的完整 rollout 语料。"],"why":"It exposes mobile post-training as auditable state/action/reward interaction while separating simulated control from real-app fidelity.","primary_link":"https://arxiv.org/abs/2605.26114","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Purewhiter/mobilegym"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Purewhiter/mobilegym/releases/tag/data-v0.1.0"},{"key":"project","label":["Project","项目主页"],"url":"https://mobilegym.github.io/"}],"link_count":5,"sections":9},{"id":"mobilellm-r1-2025","title":"MobileLLM-R1: Exploring the Limits of Sub-Billion Language Model Reasoners with Open Training Recipes","year":2026,"venue":"ICLR 2026","authors":["Changsheng Zhao","Ernie Chang","Zechun Liu","Chia-Jung Chang","Wei Wen","Chen Lai","Rick Cao","Yuandong Tian","Raghuraman Krishnamoorthi","Yangyang Shi","Vikas Chandra"],"authors_zh":"Changsheng Zhao, Ernie Chang, Zechun Liu, Chia-Jung Chang, Wei Wen, Chen Lai, Rick Cao, Yuandong Tian, Raghuraman Krishnamoorthi, Yangyang Shi, Vikas Chandra","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["general_reasoning","mathematics","code","science"],"tags":["mobile-reasoning","data-mixture","influence-estimation","capability-probing","knowledge-distillation","small-language-models","open-recipe"],"status":"partial","priority":"必读","paper_type_zh":"小模型推理的数据筛选、混合与全栈训练配方","best_for_zh":"研究推理数据配比、影响函数筛选、知识蒸馏、小模型课程与开放配方审计的读者","confidence":"high","one_line":["MobileLLM-R1 derives phase-specific mixtures with capability probes and influence estimates, then trains 140M-950M models through 4T pretraining, 200B distillation mid-training, and staged SFT without releasing the sampled corpus or full curation pipeline.","MobileLLM-R1 用 capability probe 与 influence estimate 形成分阶段数据配比，并以 4T 预训练、200B 蒸馏式中训练和分阶段 SFT 训练 140M-950M 模型，但未发布抽样语料或完整数据筛选流水线。"],"why":"It exposes concrete source weights, phase budgets, and selection signals for small-model reasoning research while showing why an open recipe and model checkpoints are not equivalent to a frozen, replayable training corpus.","primary_link":"https://arxiv.org/abs/2509.24945","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/MobileLLM-R1/tree/f518dc7e402876fc694a827385ed25de31242905"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/facebook/mobilellm-r1"}],"link_count":5,"sections":9},{"id":"mobileworld-2025","title":"MobileWorld: Benchmarking Autonomous Mobile Agents in Agent-User Interactive and MCP-Augmented Environments","year":2026,"venue":"ACL 2026 (Long Papers)","authors":["Quyu Kong","Xu Zhang","Zhenyu Yang","Nolan Gao","Chen Liu","Panrong Tong","Chenglin Cai","Hanzhang Zhou","Jianan Zhang","Liangyu Chen","Zhidan Liu","Steven Hoi","Yue Wang"],"authors_zh":"Quyu Kong, Xu Zhang, Zhenyu Yang, Nolan Gao, Chen Liu, Panrong Tong, Chenglin Cai, Hanzhang Zhou, Jianan Zhang, Liangyu Chen, Zhidan Liu, Steven Hoi, Yue Wang","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","mobile_gui_agents","tool_use"],"tags":["agent-environment","agent-trajectories","evaluation","mobile-gui","mcp","user-interaction"],"status":"partial","priority":"可读","paper_type_zh":"移动 GUI 智能体环境、混合交互基准与轨迹发布","best_for_zh":"研究移动 agent episode、跨应用长程任务、用户澄清、MCP、环境 verifier 与 replay 审计的读者","confidence":"high","one_line":["MobileWorld evaluates 201 long-horizon mobile tasks in a containerized, snapshot-reset environment with GUI, simulated-user, and MCP actions plus task-specific programmatic terminal checks.","MobileWorld 在可快照重置的容器化 Android 环境中评测 201 个长程任务，把 GUI、模拟用户澄清与 MCP 调用写入同一 episode，并以任务特定程序检查终局；当前公开轨迹可审计成功和失败，但论文运行的不可变 replay pin 仍不完整。"],"why":"It makes hybrid mobile episodes and deterministic environment feedback auditable, while exposing replay drift, verifier scope, external-tool mutability, and privacy/licensing boundaries that block unqualified training reuse.","primary_link":"https://aclanthology.org/2026.acl-long.278/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tongyi-MAI/MobileWorld"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Tongyi-MAI/MobileWorld/tree/main/site/trajs"},{"key":"project","label":["Project","项目主页"],"url":"https://tongyi-mai.github.io/MobileWorld/"}],"link_count":6,"sections":9},{"id":"modex-2026","title":"ModeX: Evaluator-Free Best-of-N Selection for Open-Ended Generation","year":2026,"venue":"ACL 2026","authors":["Hyeong Kyu Choi","Sharon Li"],"authors_zh":"Hyeong Kyu Choi、Sharon Li","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["summarization","code","mathematics","reasoning"],"tags":["best-of-n","evaluator-free-selection","spectral-clustering","modal-generation","open-ended-generation","early-pruning"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["ModeX selects an existing open-ended response by recursively isolating a dense n-gram-similarity cluster and returning its maximum-degree centroid; ModeX-Lite applies the same idea during decoding.","ModeX 通过递归分离出稠密的 n-gram 相似度簇并返回其最大度中心点，从已有的开放式回答中挑选结果；ModeX-Lite 在解码过程中应用同一思路。"],"why":"It supplies the rollout/search/test-time trace track with a deterministic selection recipe whose graph and pruning decisions can be audited separately from task scores, while modal consensus remains weaker than correctness verification.","primary_link":"https://aclanthology.org/2026.acl-long.655/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deeplearning-wisc/ModeX"}],"link_count":5,"sections":9},{"id":"multi-agent-pareto-tts-2026","title":"Multi-Agent Reasoning Improves Compute Efficiency: Pareto-Optimal Test-Time Scaling","year":2026,"venue":"ACL 2026 Student Research Workshop","authors":["Florian Valentin Wunderlich","Lars Benedikt Kaesberg","Jan Philip Wahle","Terry Ruas","Bela Gipp"],"authors_zh":"Florian Valentin Wunderlich、Lars Benedikt Kaesberg、Jan Philip Wahle、Terry Ruas、Bela Gipp（机构：哥廷根大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-scaling","multi-agent","pareto-efficiency","self-consistency","debate","mixture-of-agents"],"status":"verified","priority":"可读","paper_type_zh":"测试时扩展效率对比研究","best_for_zh":"需要在固定准确率、延迟或硬件预算下选择实际多智能体推理配置的读者。","confidence":"high","one_line":["This study shows that mixture-of-agents can dominate matched-budget single-agent scaling and gives concrete parallel-versus-sequential allocation rules.","该研究表明，混合智能体在等预算下可优于单智能体扩展，并给出并行与顺序计算分配的具体规则。"],"why":"It compares scaling mechanisms at matched cost and demonstrates that the allocation of parallel agents and sequential aggregation, not only total compute, determines efficiency.","primary_link":"https://aclanthology.org/2026.acl-srw.1/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Multi-Agent-LLMs/lm-evaluation-harness"}],"link_count":3,"sections":9},{"id":"multi-crit-pluralistic-criteria-2026","title":"Multi-Crit: Benchmarking Multimodal Judges on Pluralistic Criteria-Following","year":2026,"venue":"CVPR 2026","authors":["Tianyi Xiong","Yi Ge","Ming Li","Zuolong Zhang","Pranav Kulkarni","Kaishen Wang","Qi He","Zeying Zhu","Chenxi Liu","Ruibo Chen","Tong Zheng","Yanshuo Chen","Xiyao Wang","Renrui Zhang","Wenhu Chen","Heng Huang"],"authors_zh":"Tianyi Xiong、Yi Ge、Ming Li、Zuolong Zhang、Pranav Kulkarni、Kaishen Wang、Qi He、Zeying Zhu、Chenxi Liu、Ruibo Chen、Tong Zheng、Yanshuo Chen、Xiyao Wang、Renrui Zhang、Wenhu Chen、Heng Huang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["Multi-Crit supplies criterion-level human preferences and conflict-aware metrics for testing whether multimodal judges follow pluralistic rubrics.","Multi-Crit 提供准则级人工偏好及冲突感知指标，用于检验多模态评判模型能否遵循多元量规。"],"why":"Multi-Crit supplies criterion-level human preferences and conflict-aware metrics for testing whether multimodal judges follow pluralistic rubrics.","primary_link":"https://arxiv.org/abs/2511.21662","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/txiong23/Multi-Crit"},{"key":"project","label":["Project","项目主页"],"url":"https://multi-crit.github.io/"}],"link_count":3,"sections":9},{"id":"naturalgaia-a-verifiable-benchmark-and-hierarchical-framework-for-long-horizon-gui-tasks","title":"NaturalGAIA: A Verifiable Benchmark and Hierarchical Framework for Long-Horizon GUI Tasks","year":2026,"venue":"ACL 2026","authors":["Zihan Zheng","Tianle Cui","Taoran Wang","Fengtao Wang","Jiahui Pan","Lewei He","Qianglong Chen"],"authors_zh":"Zihan Zheng, Tianle Cui, Taoran Wang, Fengtao Wang, Jiahui Pan, Lewei He, Qianglong Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Real GUI intents are decomposed into verifiable causal pathways, with long-horizon tasks and human-verified trajectories released.","将真实 GUI 意图拆成可验证因果路径，并发布长程任务与人工核验轨迹。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2508.01330","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/KeLes-Coding/NatureGAIA"},{"key":"project","label":["Project","项目主页"],"url":"https://anonymous.4open.science/r/NatureGAIA-721F/"}],"link_count":5,"sections":9},{"id":"nvidia-nemotron-3-ultra-2026","title":"Nemotron 3 Ultra: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning","year":2026,"venue":"arXiv preprint","authors":["NVIDIA"],"authors_zh":"NVIDIA","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward","process_reward"],"training_use":["sft","distillation","rlvr","agent_training","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","science","code","software_engineering","search","agentic_tool_use","long_context","safety","instruction_following"],"tags":["nvidia","nemotron-3-ultra","frontier-report","data-disclosure-ledger","open-weights","sft","rlvr","mopd","agent-training","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型技术报告与数据披露账本","best_for_zh":"需要审计开放权重发布、混合数据来源和多环境 RL 复现边界的研究者","confidence":"high","one_line":["Nemotron 3 Ultra releases weights, data collections, and recipes while disclosing multi-domain SFT, asynchronous RLVR with 16 rollouts, and two-stage teacher distillation; private inputs, complete reward calibration, and item-level provenance remain incomplete.","Nemotron 3 Ultra 发布了权重、数据集合与训练配方，并披露多领域 SFT、每样本 16 次 rollout 的异步 RLVR 和两阶段教师蒸馏；私有输入、完整 reward 校准和逐条来源谱系仍不完整。"],"why":"It is a high-disclosure frontier report that makes the distinction between released training artifacts and non-reproducible private/vendor data, incomplete environment pins, and unresolved reward/audit boundaries unusually concrete.","primary_link":"https://arxiv.org/abs/2606.15007","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Nemotron"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/nvidia/nemotron-post-training-v3"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/"}],"link_count":6,"sections":9},{"id":"retrospective-nlg-evaluation-2026","title":"Never Truly Out of Fashion: A Retrospective Look at Evaluation in NLG","year":2026,"venue":"RetroEval 2026","authors":["Patrícia Schmidtová","Saad Mahamood","Ondřej Dušek"],"authors_zh":"Patrícia Schmidtová, Saad Mahamood, Ondřej Dušek","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"NLG 评测实践的纵向语料审计研究","best_for_zh":"需要制定 LLM judge 校准、人工评估或评测术语审计方案的研究者。","confidence":"medium","one_line":["Reviews evaluation practice and the incomplete human validation of LLM judges.","基于跨七十年论文语料，量化 NLG 评测中人工评估、LLM judge 与评测标准的长期演变。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2026.retroeval-main.8/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/patuchen/trends_in_nlg_eval"}],"link_count":2,"sections":9},{"id":"negative-aware-fine-tuning-2025","title":"NFT: Bridging Supervised Learning and Reinforcement Learning in Math Reasoning","year":2026,"venue":"ICLR 2026","authors":["Huayu Chen","Kaiwen Zheng","Qinsheng Zhang","Ganqu Cui","Lifan Yuan","Yin Cui","Haotian Ye","Tsung-Yi Lin","Ming-Yu Liu","Jun Zhu","Haoxiang Wang"],"authors_zh":"Huayu Chen；Kaiwen Zheng；Qinsheng Zhang；Ganqu Cui；Lifan Yuan；Yin Cui；Haotian Ye；Tsung-Yi Lin；Ming-Yu Liu；Jun Zhu；Haoxiang Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["negative-samples","math-reasoning","rlvr","answer-verifier","rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["NFT trains math reasoning from verifier-separated positive and negative rollouts instead of treating rejected answers as disposable.","NFT 在 DAPO-Math-17k 上进行在线训练：每个 rollout step 抽取 512 个问题、每题生成 16 个答案，并用二元正确性信号同时利用正例和负例。"],"why":"It is a concrete accepted/rejected rollout recipe, but its reuse is limited by missing verifier, rollout, and release-provenance details.","primary_link":"https://arxiv.org/abs/2505.18116","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVlabs/NFT"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/NFT-7B"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/cosmos-lab/negative-aware-ft/"}],"link_count":5,"sections":9},{"id":"wonda-invariant-curation-2026","title":"Not All Invariants Are Equal: Curating Training Data to Accelerate Program Verification with SLMs","year":2026,"venue":"ICML 2026","authors":["Ido Pinto","Yizhak Yisrael Elboher","Haoze Wu","Nina Narodytska","Guy Katz"],"authors_zh":"Ido Pinto、Yizhak Yisrael Elboher、Haoze Wu、Nina Narodytska、Guy Katz","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation","audit"],"construction_layer":["trace_writing","reward_verifier_layer","release_audit"],"domains":["program-verification","loop-invariants","C-programs"],"tags":["programmatically-verifiable-outcome","loop-invariants","program-verification","data-curation","ICML-2026"],"status":"verified","priority":"必读","paper_type_zh":"形式化 verifier 驱动的循环不变量数据构造、开放发布与 SFT 实证","best_for_zh":"构建程序验证训练数据、研究 programmatic feedback，或审计 verifier 标签与运行时归因的研究者","confidence":"high","one_line":["WONDA turns raw verifier invariants into compact loop-invariant SFT targets, but reuse requires distinguishing the 7,763-row curated table from the grade-filtered 7,284-sample experiment set and auditing lineage, licenses, and version drift.","WONDA 将 UAutomizer 原始循环不变量整理为紧凑 SFT 目标，并分别记录归纳正确性、充分性与 verifier 加速，但复用时必须区分 7,763 行 curated 表和 7,284 条实验训练集，并补查 lineage、许可与版本漂移。"],"why":"It makes a formal verifier part of the data-construction contract while revealing that a verifier-valid invariant, a useful invariant, and a faster invariant are distinct labels.","primary_link":"https://arxiv.org/abs/2603.15510","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/idopinto/wonda"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/idopinto/Wonda-Training-Dataset-Full"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/idopinto/wonda"}],"link_count":6,"sections":9},{"id":"odysseys-2026","title":"Odysseys: Benchmarking Web Agents on Realistic Long Horizon Tasks","year":2026,"venue":"arXiv preprint","authors":["Lawrence Keunho Jang","Jing Yu Koh","Daniel Fried","Ruslan Salakhutdinov"],"authors_zh":"Lawrence Keunho Jang、Jing Yu Koh、Daniel Fried、Ruslan Salakhutdinov","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","web_browsing","computer_use","long_horizon_tasks"],"tags":["long-horizon-web-agents","computer-use","live-web","osworld","human-browsing-journeys","rubric-evaluation","llm-as-judge","partial-credit","trajectory-efficiency","success-failure-retention","environment-drift","replay-risk","privacy","security"],"status":"partial","priority":"必读","paper_type_zh":"长程实时网页智能体评测基准与 rubric 数据发布","best_for_zh":"研究长程 computer-use、细粒度轨迹评估、LLM judge、实时网页重放与环境审计的读者","confidence":"high","one_line":["Odysseys releases 200 live-web tasks with 1,225 rubric checkpoints and an OSWorld/Gemini scoring pipeline, but not the 2,380 source histories, complete model trajectories, human labels, frozen sites, or deterministic replay state.","Odysseys 公开 200 个长程实时网页任务、1,225 条 rubric 和 200 份 OSWorld 配置及评分脚本，但没有公开来源浏览历史、完整模型轨迹、人工标签或可确定重放的网页状态。"],"why":"It replaces one holistic pass/fail judgment with checkpoint-level episode feedback over screenshots and actions, exposing partial progress and efficiency on multi-site tasks while making environment drift, judge dependence, incomplete- run handling, and absent rollout lineage explicit reuse risks.","primary_link":"https://arxiv.org/abs/2604.24964","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ljang0/Odysseys"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ljang0/Odysseys/blob/main/data/odysseys.json"},{"key":"project","label":["Project","项目主页"],"url":"https://odysseys-website.pages.dev/"}],"link_count":11,"sections":9},{"id":"mmpo-marginal-likelihood-preference-2026","title":"Offline Preference Optimization via Maximum Marginal Likelihood Estimation","year":2026,"venue":"EACL 2026 Long Papers","authors":["Saeed Najafi","Alona Fyshe"],"authors_zh":"Saeed Najafi, Alona Fyshe（阿尔伯塔大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","maximum-marginal-likelihood","dpo","offline-alignment"],"status":"verified","priority":"可读","paper_type_zh":"离线偏好优化目标研究","best_for_zh":"适合比较 DPO 类目标、希望稳定使用既有偏好数据集且不训练显式奖励模型的读者。","confidence":"high","one_line":["MMPO converts each offline preference pair into a normalized maximum-marginal-likelihood objective that implicitly up-weights the chosen response.","MMPO 将每个离线偏好回答对转化为归一化的最大边缘似然目标，并隐式提高优选回答的权重。"],"why":"It changes the optimization consumer of an unchanged preference dataset and empirically studies the trade-off between alignment gain and retained general capability.","primary_link":"https://aclanthology.org/2026.eacl-long.318/","links":[],"link_count":2,"sections":9},{"id":"omni-reward-generalist-omni-modal-reward-modeling-2026","title":"Omni-Reward: Towards Generalist Omni-Modal Reward Modeling with Free-Form Preferences","year":2026,"venue":"ICLR 2026","authors":["Zhuoran Jin","Hongbang Yuan","Kejian Zhu","Jiachun Li","Pengfei Cao","Yubo Chen","Kang Liu","Jun Zhao"],"authors_zh":"Zhuoran Jin、Hongbang Yuan、Kejian Zhu 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["reward_modeling","preference_learning","evaluation"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["multimodal-generation","reward-modeling"],"tags":["candidate-batch","preference-data","multimodal","reward-model"],"status":"verified","priority":"必读","paper_type_zh":"多模态偏好数据集、奖励模型与偏好评测","best_for_zh":"需要构建跨模态奖励模型，或审计自由文本偏好准则的研究者。","confidence":"high","one_line":["Omni-RewardData provides multimodal preference pairs and free-form criteria for generalist reward modeling.","Omni-RewardData 将跨五种模态的候选比较与自由文本偏好准则一起公开，使通用奖励模型能够按指定判断维度评分。"],"why":"It makes modality and preference criterion part of the reward record instead of treating all pairwise labels as interchangeable.","primary_link":"https://arxiv.org/abs/2510.23451","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/OmniReward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jinzhuoran/OmniRewardData"},{"key":"project","label":["Project","项目主页"],"url":"https://omnireward.github.io/"}],"link_count":5,"sections":9},{"id":"omni-rrm-advancing-omni-reward-modeling-via-automatic-rubric-grounded-preference-synthesis-2026","title":"Omni-RRM: Advancing Omni Reward Modeling via Automatic Rubric-Grounded Preference Synthesis","year":2026,"venue":"arXiv","authors":["Zicheng Kong","Dehua Ma","Zhenbo Xu","Alven Yang","Yiwei Ru","Haoran Wang","Zixuan Zhou","Fuqing Bie","Liuyu Xiang","Huijia Wu","Jian Zhao","Zhaofeng He"],"authors_zh":"Zicheng Kong、Dehua Ma、Zhenbo Xu、Alven Yang、Yiwei Ru 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","Omni-RRM 自动合成带模态量规和逐维理由的跨文本、图像、视频、音频偏好数据，训练全模态奖励模型。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2602.00846","links":[{"key":"code","label":["Code","代码"],"url":"https://anonymous.4open.science/r/Omni-RRM-CC08"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Omni-RRM/Omni-Preference"}],"link_count":3,"sections":9},{"id":"terminal-capabilities-data-engineering-2026","title":"On Data Engineering for Scaling LLM Terminal Capabilities","year":2026,"venue":"arXiv preprint","authors":["Renjie Pi","Grace Lam","Mohammad Shoeybi","Pooya Jannaty","Bryan Catanzaro","Wei Ping"],"authors_zh":"Renjie Pi、Grace Lam、Mohammad Shoeybi、Pooya Jannaty、Bryan Catanzaro、Wei Ping","tracks":["rollout_search_test_time_trace_data","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["terminal_agents","shell_and_cli","docker_environments","agent_trajectories","math","code","software_engineering","data_processing","data_querying","data_science","debugging","dependency_management","file_operations","scientific_computing","security","system_administration"],"tags":["nemotron-terminal","terminal-corpus","terminal-task-gen","terminal-agents","terminus-2","docker","harbor","agent-trajectories","synthetic-data","dataset-adapters","failed-trajectories","sft","scaling-study","programmatic-verification","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"终端 agent 轨迹数据发布、构建配方与规模研究","best_for_zh":"关注终端 agent SFT、rollout/失败轨迹、程序化验证、容器环境与开放数据审计的研究者","confidence":"medium","one_line":["Releases exactly 366,154 train-only terminal SFT episodes from dataset adapters and skill-based synthesis, while the paper's no-filter recipe retains incomplete and failed behavior but the public rows expose no normalized outcome or reward.","公开 366,154 条 train-only 终端 SFT episode；论文研究的完整混合共 490,520 条，其中 124,366 条 seed-based 轨迹尚未确认发布，且公开行没有规范化 outcome、reward 或 test result。"],"why":"It provides unusually direct evidence that retaining realistic error states can outperform success-only filtering for terminal-agent SFT, while making count reconciliation, failure labels, verifier semantics, and container replayability first-class audit requirements.","primary_link":"https://arxiv.org/abs/2602.21193","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-Terminal-Corpus"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/nemotron-terminal"}],"link_count":12,"sections":9},{"id":"vlm-test-time-scaling-2026","title":"On Test-Time Scaling for Vision-Language Models","year":2026,"venue":"arXiv preprint","authors":["Fawaz Sammani","Tzoulio Chamiti","Nikos Deligiannis"],"authors_zh":"Fawaz Sammani、Tzoulio Chamiti、Nikos Deligiannis（机构：布鲁塞尔自由大学、imec）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["multimodal-reasoning","mathematical-reasoning","visual-question-answering"],"tags":["test-time-scaling","vision-language-models","self-consistency","overthinking","multimodal-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多模态测试时扩展基准与分析","best_for_zh":"需要在视觉语言模型尺寸与推理策略之间分配推理预算的读者。","confidence":"high","one_line":["A broad LVLM study finds that standard test-time scaling can strongly help capable small models but can hurt perception tasks when extra reasoning makes models lose visual focus.","这项大规模视觉语言模型研究发现，标准测试时扩展能显著帮助有能力的小模型，但额外推理也可能使模型失去视觉焦点并损害感知任务。"],"why":"It extends test-time scaling evaluation beyond text-only reasoning and identifies when extra computation improves visual reasoning versus induces overthinking.","primary_link":"https://arxiv.org/abs/2606.28864","links":[],"link_count":1,"sections":9},{"id":"fragility-benchmark-contamination-detection-2025","title":"On the Fragility of Benchmark Contamination Detection in Reasoning Models","year":2026,"venue":"ICLR 2026","authors":["Han Wang","Haoyu Li","Brian Ko","Huan Zhang"],"authors_zh":"Han Wang, Haoyu Li, Brian Ko, Huan Zhang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","benchmark-contamination","iclr-2026"],"status":"verified","priority":"可读","paper_type_zh":"推理模型基准污染检测脆弱性审计论文","best_for_zh":"需要审计推理模型训练污染和检测器失效的研究者。","confidence":"high","one_line":["PPO-style RL can erase contamination-detection signals after contaminated SFT, while CoT contamination of mature LRMs drives existing detectors toward chance.","PPO 类强化学习可在污染 SFT 后抹去检测信号，而成熟推理模型的 CoT 污染会让既有检测器接近随机猜测。"],"why":"It demonstrates that conventional contamination detectors can fail exactly in practical reasoning-model training pipelines.","primary_link":"https://openreview.net/forum?id=bhR00j6Mku","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ASTRAL-Group/LRM_Conta_Detection_Arena"}],"link_count":3,"sections":9},{"id":"ai-reviewers-nature-expert-scientists-2026","title":"On the limits and opportunities of AI reviewers: Reviewing the reviews of Nature-family papers with 45 expert scientists","year":2026,"venue":"arXiv preprint","authors":["Seungone Kim","Dongkeun Yoon","Kiril Gashteovski","Juyoung Suk","Jinheon Baek","Pranjal Aggarwal","Ian Wu","Viktor Zaverkin","Spase Petkoski","Daniel R. Schrider","Ilija Dukovski","Francesco Santini","Biljana Mitreska","Yong Jeong","Kyeongha Kwon","Young Min Sim","Dragana Manasova","Arthur Porto","Biljana Mojsoska","Makoto Takamoto","Marko Shuntov","Ruoqi Liu","Hyunjoo Jenny Lee","Niyazi Ulas Dinç","Yehhyun Jo","Sunkyu Han","Chungwoo Lee","Huishan Li","Esther H. R. Tsai","Ergun Simsek","Khushboo Shafi","Yeonseung Chung","Jihye Park","Aleksandar Shulevski","Henrik Christiansen","Yoosang Son","Elly Knight","Amanda Montoya","Jeongyoun Ahn","Christian Langkammer","Heera Moon","Changwon Yoon","Nikola Stikov","Mooseok Jang","Edward Choi","Junhan Kim","Yeon Sik Jung","Woo Youn Kim","Jae Kyoung Kim","Ishraq Md Anjum","Hyun Uk Kim","Drew Bridges","Carolin Lawrence","Xiang Yue","Alice Oh","Akari Asai","Sean Welleck","Graham Neubig"],"authors_zh":"Seungone Kim、Dongkeun Yoon、Kiril Gashteovski、Juyoung Suk、Jinheon Baek、Pranjal Aggarwal、Ian Wu、Viktor Zaverkin、Spase Petkoski、Daniel R. Schrider、Ilija Dukovski、Francesco Santini、Biljana Mitreska、Yong Jeong、Kyeongha Kwon、Young Min Sim、Dragana Manasova、Arthur Porto、Biljana Mojsoska、Makoto Takamoto、Marko Shuntov、Ruoqi Liu、Hyunjoo Jenny Lee、Niyazi Ulas Dinç、Yehhyun Jo、Sunkyu Han、Chungwoo Lee、Huishan Li、Esther H. R. Tsai、Ergun Simsek、Khushboo Shafi、Yeonseung Chung、Jihye Park、Aleksandar Shulevski、Henrik Christiansen、Yoosang Son、Elly Knight、Amanda Montoya、Jeongyoun Ahn、Christian Langkammer、Heera Moon、Changwon Yoon、Nikola Stikov、Mooseok Jang、Edward Choi、Junhan Kim、Yeon Sik Jung、Woo Youn Kim、Jae Kyoung Kim、Ishraq Md Anjum、Hyun Uk Kim、Drew Bridges、Carolin Lawrence、Xiang Yue、Alice Oh、Akari Asai、Sean Welleck、Graham Neubig","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["PeerReviewBench uses expert scientists to rate individual human and AI review criticisms for correctness, significance, and evidential sufficiency.","PeerReviewBench 由领域科学家逐条评定人类与 AI 评审意见的正确性、重要性和证据充分性。"],"why":"PeerReviewBench uses expert scientists to rate individual human and AI review criticisms for correctness, significance, and evidential sufficiency.","primary_link":"https://arxiv.org/abs/2605.20668","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/prometheus-eval/peerreview-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/prometheus-eval/peerreview-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://paperreview.ai/"}],"link_count":4,"sections":9},{"id":"shelf-life-llm-judges-2026","title":"On the Shelf Life of Finetuned LLM-Judges: Future Proofing, Backward Compatibility, and Question Generalization","year":2026,"venue":"ICLR 2026","authors":["Janvijay Singh","Austin Xu","Yilun Zhou","Yefan Zhou","Dilek Hakkani-Tür","Shafiq Joty"],"authors_zh":"Janvijay Singh，Austin Xu，Yilun Zhou，Yefan Zhou，Dilek Hakkani-Tür，Shafiq Joty。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"微调 LLM judge 的分布漂移与泛化可靠性审计","best_for_zh":"训练、更新或部署微调 pairwise judge 的团队。","confidence":"medium","one_line":["Audits whether a finetuned judge remains reliable as questions and model generations shift.","审计微调 judge 面对新旧生成器与未见问题时的保鲜期，并表明持续训练更平衡。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://openreview.net/forum?id=hzah1nToLx","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/iamjanvijay/judge-training-analysis"},{"key":"project","label":["Project","项目主页"],"url":"https://iamjanvijay.github.io/judge-training-analysis"}],"link_count":5,"sections":9},{"id":"step-length-confounding-selection-2026","title":"On the Step Length Confounding in LLM Reasoning Data Selection","year":2026,"venue":"Findings of ACL 2026","authors":["Bing Wang","Rui Miao","Chen Shen","Shaotian Yan","Kaiyuan Liu","Ximing Li","Xiaosong Yuan","Sinan Fan","Jun Zhang","Jieping Ye"],"authors_zh":"Bing Wang, Rui Miao, Chen Shen, Shaotian Yan, Kaiyuan Liu, Ximing Li, Xiaosong Yuan, Sinan Fan, Jun Zhang, Jieping Ye","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","audit_failure","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft"],"construction_layer":["trace_writing","optimizer_scaffold","release_audit"],"domains":["reasoning","mathematics","science"],"tags":["reasoning-data-selection","naturalness","debiasing","sft"],"status":"verified","priority":"必读","paper_type_zh":"推理数据选择审计、去偏方法与数据发布","best_for_zh":"用似然或自然性分数整理多步骤推理监督微调轨迹的读者。","confidence":"high","one_line":["ASLEC removes step-length bias from likelihood-based reasoning-data selection by dropping or regressing out first-token effects.","ASLEC 通过丢弃或回归掉首词元效应，消除基于似然的推理数据选择中的步骤长度偏差。"],"why":"It exposes how a formatting-dependent denominator can silently change which reasoning traces are trained on.","primary_link":"https://aclanthology.org/2026.findings-acl.918/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wangbing1416/ASLEC"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/wangbing1416/msms-cot-sft"}],"link_count":5,"sections":9},{"id":"open-swe-traces-2026","title":"Open-SWE-Traces: Advancing Dual-Mode Multilingual Distillation for Software Engineering Agents","year":2026,"venue":"arXiv","authors":["Wasi Uddin Ahmad","Nikolai Ludwig","Somshubra Majumdar","Boris Ginsburg"],"authors_zh":"Wasi Uddin Ahmad、Nikolai Ludwig、Somshubra Majumdar、Boris Ginsburg","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","agent_environment"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["full_episode","state_action_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["trace_writing","search_substrate","release_audit"],"domains":["software_engineering","repository_level_code","multilingual_code"],"tags":["open-swe-traces","swe-agent","openhands","software-engineering","trajectory-distillation","multilingual-code","failed-trajectories"],"status":"verified","priority":"必读","paper_type_zh":"多语言软件工程 agent 轨迹数据发布与双模式蒸馏配方","best_for_zh":"关注 repository-level agent、SWE-bench 类环境、工具交互轨迹、失败数据利用和可执行反馈审计的研究者","confidence":"high","one_line":["Releases 207,489 multilingual software-agent episodes from 20,000 SWE-rebench-V2 pull-request tasks across two teachers and two harness families, retaining solved, unresolved, and unavailable outcomes.","发布 207,489 条多语言软件工程 agent episode，来源为 20,000 个 SWE-rebench-V2 PR 任务、两类教师模型与两类 harness，并保留已解决、未解决和结果不可用三类 outcome。"],"why":"It makes large-scale SWE-agent distillation inspectable at episode level and provides direct evidence about the value and risks of training on non-success trajectories.","primary_link":"https://arxiv.org/abs/2606.16038","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Open-SWE-Traces"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/open-swe-traces"}],"link_count":5,"sections":9},{"id":"open-discovery-trace-ai-scientist-2026","title":"OpenDiscoveryTrace: Process Traces for Evaluating AI Scientist Workflows","year":2026,"venue":"ICML 2026 AI for Science Workshop Dataset Proposal Competition","authors":["Anonymous"],"authors_zh":"匿名作者（公开版本未披露）","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["ai-scientist-agents","process-traces","error-localization"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究 AI 科学家过程评测、错误定位和工作流质量建模的读者。","confidence":"high","one_line":["OpenDiscoveryTrace releases complete AI-scientist workflows—thoughts, tools, errors, revisions, and outcomes—for process-level evaluation.","OpenDiscoveryTrace 开放 AI 科学家从思考、工具调用到错误修正和结论的完整工作流轨迹。"],"why":"It makes scientific-agent failures observable at the trace level rather than only scoring final claims.","primary_link":"https://openreview.net/forum?id=EHT3wVhCUZ","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/aayambansal/OpenDiscoveryTrace"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/aayambansall/OpenDiscoveryTrace"}],"link_count":3,"sections":9},{"id":"opengenalign-2026","title":"OpenGenAlign: A Preference Dataset and Benchmark for Trustworthy Reward Modeling in Open-Ended, Long-Context Generation","year":2026,"venue":"Findings of ACL 2026","authors":["Hanning Zhang","Juntong Song","Juno Zhu","Yuanhao Wu","Tong Zhang","Cheng Niu"],"authors_zh":"Hanning Zhang, Juntong Song, Juno Zhu, Yuanhao Wu, Tong Zhang, Cheng Niu","tracks":["judgment_rubric_domain_expert_data","preference_reward_feedback_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"长上下文偏好数据、奖励模型与策略优化论文","best_for_zh":"需要评估或训练长问答、数据到文本、摘要奖励模型的研究者。","confidence":"medium","one_line":["A 33K preference dataset and benchmark for reward modeling of long-context open-ended generation.","用多次 o3 判决构造长上下文偏好数据，专门衡量幻觉与完整性等生成质量。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.findings-acl.553/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hanningzhang/OpenGenAlign"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HanningZhang/OpenGenAlign"}],"link_count":5,"sections":9},{"id":"openmmreasoner-2025","title":"OpenMMReasoner: Pushing the Frontiers for Multimodal Reasoning with an Open and General Recipe","year":2026,"venue":"CVPR 2026","authors":["Kaichen Zhang","Keming Wu","Zuhao Yang","Bo Li","Kairui Hu","Bin Wang","Ziwei Liu","Xingxuan Li","Lidong Bing"],"authors_zh":"Kaichen Zhang、Keming Wu、Zuhao Yang、Bo Li、Kairui Hu、Bin Wang、Ziwei Liu、Xingxuan Li、Lidong Bing","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["multimodal","visual_reasoning","mathematics","science","general_reasoning"],"tags":["openmmreasoner","multimodal-reasoning","visual-reasoning","cold-start-sft","online-rollouts","gspo","mixed-reward","llm-judge","partial-lineage"],"status":"verified","priority":"必读","paper_type_zh":"多模态推理数据发布与SFT-to-RL配方","best_for_zh":"视觉推理冷启动、RLVR、混合奖励与开放发布研究者","confidence":"high","one_line":["Separates an open 874K multimodal SFT conversation mixture from roughly 73K released RL training prompts and unreleased 16-way online GSPO rollouts, with mixed rule, learned-judge, and format rewards.","公开约87.4万条多模态SFT对话与72,971条RL训练提示，训练时再在线采样16路回答并用规则/裁判正确性与格式奖励驱动GSPO，但rollout本身未发布。"],"why":"It makes a two-stage multimodal reasoning recipe inspectable and shows why released prompts, generated trajectories, terminal rewards, and policy updates must remain distinct audit objects.","primary_link":"https://arxiv.org/abs/2511.16334","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EvolvingLMMs-Lab/OpenMMReasoner"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/lmms-lab/openmmreasoner"},{"key":"project","label":["Project","项目主页"],"url":"https://evolvinglmms-lab.github.io/OpenMMReasoner/"}],"link_count":6,"sections":9},{"id":"openmobile-task-trajectory-synthesis-2026","title":"OpenMobile: Building Open Mobile Agents with Task and Trajectory Synthesis","year":2026,"venue":"COLM 2026","authors":["Kanzhi Cheng","Zehao Li","Zheng Ma","Nuo Chen","Jialin Cao","Qiushi Sun","Zichen Ding","Fangzhi Xu","Hang Yan","Jiajun Chen","Luu Anh Tuan","Jianbing Zhang","Lewei Lu","Dahua Lin"],"authors_zh":"Kanzhi Cheng, Zehao Li, Zheng Ma, Nuo Chen, Jialin Cao, Qiushi Sun, Zichen Ding, Fangzhi Xu, Hang Yan, Jiajun Chen, Luu Anh Tuan, Jianbing Zhang, Lewei Lu, Dahua Lin","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mobile-agents","computer-use","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"移动端 GUI 智能体任务合成、错误恢复轨迹与训练数据论文","best_for_zh":"需要开放移动端任务、执行轨迹和错误恢复监督来训练或研究手机智能体的研究者。","confidence":"high","one_line":["OpenMobile releases synthesized Android tasks and policy-switching mobile-agent trajectories that retain error-recovery signals.","OpenMobile 以环境记忆合成移动端任务，并用强弱策略切换保留错误恢复过程，公开任务与轨迹联合训练数据。"],"why":"It jointly opens task synthesis, recovery-rich execution traces, training splits, and the construction code for mobile agents.","primary_link":"https://arxiv.org/abs/2604.15093","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/njucckevin/OpenMobile-Code"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/cckevinn/OpenMobile-Data"}],"link_count":5,"sections":9},{"id":"openresearcher-dataset-2026","title":"OpenResearcher: A Fully Open Pipeline for Long-Horizon Deep Research Trajectory Synthesis","year":2026,"venue":"arXiv preprint","authors":["Li, Zhuofeng","Jiang, Dongfu","Ma, Xueguang","Zhang, Haoxiang","Nie, Ping","Zhang, Yuyu","Zou, Kai","Xie, Jianwen","Zhang, Yu","Chen, Wenhu"],"authors_zh":"Li, Zhuofeng、Jiang, Dongfu、Ma, Xueguang、Zhang, Haoxiang、Nie, Ping、Zhang, Yuyu、Zou, Kai、Xie, Jianwen、Zhang, Yu、Chen, Wenhu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["offline-long-horizon-research-trajectories"],"tags":["instruction-demonstration-rationale","arxiv-2603.20278","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"深度研究代理监督微调","confidence":"high","one_line":["OpenResearcher bootstraps an offline evidence corpus, then generates and filters complete search-browse-answer trajectories without repeated live-web dependence.","OpenResearcher 先建立离线证据库，再合成并筛选 9.7 万多条搜索、浏览与回答交错的长程研究轨迹。"],"why":"Online research-trajectory synthesis is expensive and unstable because failed searches may reflect either a weak agent or missing documents.","primary_link":"https://arxiv.org/abs/2603.20278","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenResearcher/OpenResearcher-Dataset"}],"link_count":2,"sections":9},{"id":"openrubrics-2026","title":"OpenRubrics: Towards Scalable Synthetic Rubric Generation for Reward Modeling and LLM Alignment","year":2026,"venue":"ACL 2026","authors":["Tianci Liu","Ran Xu","Tony Yu","Ilgee Hong","Carl Yang","Tuo Zhao","Haoyu Wang"],"authors_zh":"Tianci Liu, Ran Xu, Tony Yu, Ilgee Hong, Carl Yang, Tuo Zhao, Haoyu Wang","tracks":["judgment_rubric_domain_expert_data","preference_reward_feedback_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"rubric 生成、偏好数据与奖励模型论文","best_for_zh":"需要构造可解释成对奖励信号，或审计 rubric 是否忠实反映偏好的研究者。","confidence":"high","one_line":["An open release of generated rubrics and preference data for rubric-conditioned reward modeling.","用对比偏好生成并验证 rubric，把隐式偏好转为可训练的显式奖励依据。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.acl-long.791/","links":[{"key":"code","label":["Code","代码"],"url":"https://huggingface.co/OpenRubrics"}],"link_count":4,"sections":9},{"id":"openwebrl-2026","title":"OpenWebRL: Demystifying Online Multi-turn Reinforcement Learning for Visual Web Agents","year":2026,"venue":"arXiv preprint","authors":["Rui Yang","Qianhui Wu","Yuxi Chen","Hao Bai","Wenlin Yao","Hao Cheng","Baolin Peng","Huan Zhang","Tong Zhang","Jianfeng Gao"],"authors_zh":"Rui Yang, Qianhui Wu, Yuxi Chen, Hao Bai, Wenlin Yao, Hao Cheng, Baolin Peng, Huan Zhang, Tong Zhang, Jianfeng Gao","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","agent_environment","verifier_reward","construction_recipe"],"verification_contract":["programmatic","environmental","judgment_required"],"supervision_granularity":["step_level","full_episode","scalar_reward"],"training_use":["sft","distillation","reward_modeling","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","environment_interaction","web_agents","browser_automation","multimodal_agents"],"tags":["environment-agent-trajectory-data","visual-web-agent","browser-agent","online-reinforcement-learning","multimodal-trajectory","trajectory-level-reward","vlm-judge","sft-trajectories","live-web","verifier-audit"],"status":"partial","priority":"必读","paper_type_zh":"视觉Web智能体在线多轮强化学习框架与轨迹/任务/judge数据发布","best_for_zh":"研究Web智能体SFT、online RL、全轨迹奖励、VLM judge与live-web复现审计的读者","confidence":"medium","one_line":["OpenWebRL releases 412 success-only SFT trajectories, 2,198 RL task specifications, live-browser MM-GRPO code, hybrid trajectory judging and model weights, while its roughly 54K online training rollouts and exact live-web replay remain unavailable.","OpenWebRL公开412条success-only SFT轨迹、2,198个RL task spec、hybrid trajectory reward与训练框架，但约54K online rollout未确认发布，2,102默认快照、live-web replay及Judge-13K权利仍需单独审计。"],"why":"It makes the full causal chain from browser task and observation through actions, environment feedback, terminal judgment and policy optimization inspectable, while showing why task prompts, selected demonstrations and runtime online rollouts must be tracked as different data objects with different success balance, replay and rights risks.","primary_link":"https://arxiv.org/abs/2606.02031v2","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenWebRL/OpenWebRL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenWebRL/OpenWebRL-SFT-Trajectories"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/OpenWebRL"},{"key":"project","label":["Project","项目主页"],"url":"https://openwebrl.github.io/"}],"link_count":6,"sections":9},{"id":"optimal-llm-prm-aggregation-2026","title":"Optimal Aggregation of LLM and PRM Signals for Efficient Test-Time Scaling","year":2026,"venue":"ICLR 2026","authors":["Peng Kuang","Yanli Wang","Xiaoyu Han","Yaowenqi Liu","Kaidi Xu","Haohan Wang"],"authors_zh":"Peng Kuang、Yanli Wang、Xiaoyu Han、Yaowenqi Liu、Kaidi Xu、Haohan Wang","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","scaling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配的读者。","confidence":"high","one_line":["Optimal Aggregation of LLM and PRM Signals for Efficient Test-Time Scaling","当过程奖励模型分数与生成器频率产生相关错误时，它可能损害选择。"],"why":"It measures a concrete decision about additional inference computation.","primary_link":"https://arxiv.org/abs/2510.13918","links":[],"link_count":3,"sections":9},{"id":"optimal-self-consistency-efficient-reasoning-llms-2025","title":"Optimal Self-Consistency for Efficient Reasoning with Large Language Models","year":2026,"venue":"ICML 2026","authors":["Austin Feng","Marius Alonso","Ambroise Odonnat","Vasilii Feofanov","Ievgen Redko"],"authors_zh":"Austin Feng；Marius Alonso；Ambroise Odonnat；Vasilii Feofanov；Ievgen Redko","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["mathematics","reasoning"],"tags":["self-consistency","blend-asc","adaptive-allocation","test-time-compute","majority-vote"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A global self-consistency sampler that reallocates chain-of-thought calls toward uncertain questions without updating the model.","Blend-ASC 在全局样本预算下按答案计数置信度把下一次 CoT 采样分给最不确定的问题，最后以经验众数输出答案。"],"why":"It makes repeated-answer counts and allocation history central Track 5 data objects while clearly separating inference-time scaling from post-training data construction.","primary_link":"https://arxiv.org/abs/2511.12309","links":[],"link_count":2,"sections":9},{"id":"orca-audio-qa-correctness-2026","title":"ORCA: Open-ended Response Correctness Assessment for Audio Question Answering","year":2026,"venue":"TACL 2026","authors":["Šimon Sedláček","Sara Barahona","Cecilia Bolaños","Laura Herrera-Alarcón","Sathvik Udupa","Fernando López","Allison Ferner","Bolaji Yusuf","Alicia Lozano-Diez","Santosh Kesiraju","Ramani Duraiswami","Jan Černocký"],"authors_zh":"Šimon Sedláček、Sara Barahona、Cecilia Bolaños、Laura Herrera-Alarcón、Sathvik Udupa、Fernando López、Allison Ferner、Bolaji Yusuf、Alicia Lozano-Diez、Santosh Kesiraju、Ramani Duraiswami、Jan Černocký","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["ORCA models open-ended audio-QA correctness as a distribution, capturing both scores and human disagreement.","把开放式音频问答的正确性建模为分布，以同时刻画评分与人类分歧。"],"why":"ORCA models open-ended audio-QA correctness as a distribution, capturing both scores and human disagreement.","primary_link":"https://arxiv.org/abs/2512.09066","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/BUT-FIT/orca-audio-qa-annotations"}],"link_count":2,"sections":9},{"id":"rationalerm-reasoning-process-2026","title":"Outcome Accuracy is Not Enough: Aligning the Reasoning Process of Reward Models","year":2026,"venue":"arXiv preprint","authors":["Binghai Wang","Yantao Liu","Yuxuan Liu","Tianyi Tang","Shenzhi Wang","Chang Gao","Chujie Zheng","Yichang Zhang","Le Yu","Shixuan Liu","Tao Gui","Qi Zhang","Xuanjing Huang","Bowen Yu","Fei Huang","Junyang Lin"],"authors_zh":"Binghai Wang、Yantao Liu、Yuxuan Liu、Tianyi Tang、Shenzhi Wang、Chang Gao、Chujie Zheng、Yichang Zhang、Le Yu、Shixuan Liu、Tao Gui、Qi Zhang、Xuanjing Huang、Bowen Yu、Fei Huang、Junyang Lin","tracks":["training_usage_optimization_objectives","preference_reward_feedback_data"],"source_role":["process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["process_reward"],"training_use":["reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["instruction-tuning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"奖励模型理由一致性与混合奖励训练论文","best_for_zh":"需要检查奖励模型是否以正确理由作出判断、并把该检查用于强化学习的读者。","confidence":"high","one_line":["RationaleRM measures whether a reward model’s rationale agrees with atomic human reasons and uses that signal alongside outcome accuracy for RL.","RationaleRM 衡量奖励模型的理由是否与原子化人工理由一致，并将该信号与结果准确率共同用于强化学习。"],"why":"It makes the connection between a data object and a training objective inspectable.","primary_link":"https://arxiv.org/abs/2602.04649","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/RationaleRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Qwen/RationaleRM"}],"link_count":3,"sections":9},{"id":"pacore-parallel-coordinated-reasoning-2026","title":"PaCoRe: Learning to Scale Test-Time Compute with Parallel Coordinated Reasoning","year":2026,"venue":"ACL 2026 Long Papers","authors":["Jingcheng Hu","Yinmin Zhang","Shijie Shang","Xiaobo Yang","Yue Peng","Zhewei Huang","Hebin Zhou","Xin Wu","Jie Cheng","Fanqi Wan","Xiangwen Kong","Chengyuan Yao","Kaiwen Yan","Ailin Huang","Hongyu Zhou","Qi Han","Zheng Ge","Daxin Jiang","Xiangyu Zhang","Heung-Yeung Shum"],"authors_zh":"Jingcheng Hu、Heung-Yeung Shum（清华大学）；Yinmin Zhang、Shijie Shang、Xiaobo Yang、Yue Peng、Zhewei Huang、Hebin Zhou、Xin Wu、Jie Cheng、Fanqi Wan、Xiangwen Kong、Chengyuan Yao、Kaiwen Yan、Ailin Huang、Hongyu Zhou、Qi Han、Zheng Ge、Daxin Jiang、Xiangyu Zhang（StepFun）；Xiaobo Yang（北京大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["rlvr","test_time_compute","sft","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code-generation","scientific-reasoning"],"tags":["test-time-compute","parallel-reasoning","message-passing","rlvr","context-window"],"status":"verified","priority":"必读","paper_type_zh":"并行协同测试时扩展与强化学习研究","best_for_zh":"研究超越顺序上下文长度限制的大预算推理系统的读者。","confidence":"high","one_line":["PaCoRe coordinates large parallel trajectory pools through compact messages, allowing multi-million-token effective reasoning without exceeding a model’s context window.","PaCoRe 用多轮并行轨迹与压缩消息协调推理，在不超出上下文窗口的情况下扩展到数百万级有效测试时令牌。"],"why":"It makes cross-trajectory synthesis a learned test-time capability and reports scaling curves at explicitly stated token budgets.","primary_link":"https://aclanthology.org/2026.acl-long.1253/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/stepfun-ai/PaCoRe"}],"link_count":3,"sections":9},{"id":"latent-parallel-tts-2026","title":"Parallel Test-Time Scaling for Latent Reasoning Models","year":2026,"venue":"ACL 2026","authors":["Runyang You","Yongqi Li","Meng Liu","Wenjie Wang","Liqiang Nie","Wenjie Li"],"authors_zh":"Runyang You、Yongqi Li、Meng Liu、Wenjie Wang、Liqiang Nie、Wenjie Li（机构：香港理工大学、山东建筑大学、中国科学技术大学、哈尔滨工业大学深圳）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-scaling","latent-reasoning","parallel-sampling","latent-reward-model","uncertainty"],"status":"verified","priority":"可读","paper_type_zh":"潜式推理测试时扩展研究","best_for_zh":"希望把 pass@k 式推理扩展迁移到连续推理模型、而这些模型没有词元概率或文本过程奖励输入的读者。","confidence":"high","one_line":["This work brings parallel sampling and reward-guided selection to latent reasoning by using dropout or noise for exploration and a step-level latent reward model for aggregation.","该工作用 dropout 或噪声在潜空间探索，并用逐步潜空间奖励模型聚合，把并行采样与奖励引导选择带入潜式推理。"],"why":"It shows that inference scaling is not tied to textual chain-of-thought and supplies the sampling and scoring primitives needed to allocate compute in latent space.","primary_link":"https://aclanthology.org/2026.acl-long.2069/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ModalityDance/LatentTTS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/gsm8k"}],"link_count":4,"sections":9},{"id":"multi-sequence-verifier-2026","title":"Parallel Test-Time Scaling with Multi-Sequence Verifiers","year":2026,"venue":"arXiv preprint","authors":["Yegon Kim","Seungyoo Lee","Chaeyun Jang","Hyungi Lee","Juho Lee"],"authors_zh":"Yegon Kim、Seungyoo Lee、Chaeyun Jang、Hyungi Lee、Juho Lee（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","verifier","best-of-n","early-stopping"],"status":"verified","priority":"必读","paper_type_zh":"集合级验证器与并行测试时扩展论文","best_for_zh":"优化并行候选选择和时延预算的读者。","confidence":"high","one_line":["MSV jointly scores parallel candidates to improve best-of-N selection and reduce latency through streaming early stopping.","MSV 联合为并行候选打分，以改善 best-of-N 选择并通过流式提前停止降低时延。"],"why":"It treats interactions among candidates as evidence for verifier scaling.","primary_link":"https://arxiv.org/abs/2603.03417","links":[],"link_count":2,"sections":9},{"id":"pat-planning-after-trial-2026","title":"PaT: Planning-after-Trial for Efficient Test-Time Code Generation","year":2026,"venue":"ACL 2026 Long Papers","authors":["Youngsik Yoon","Sungjae Lee","Seockbean Song","Siwei Wang","Wei Chen","Jungseul Ok"],"authors_zh":"Youngsik Yoon、Sungjae Lee、Seockbean Song、Jungseul Ok（浦项科技大学）；Siwei Wang、Wei Chen（微软亚洲研究院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["code-generation"],"tags":["test-time-compute","code-generation","planning","verification","adaptive-inference"],"status":"verified","priority":"可读","paper_type_zh":"验证失败触发的自适应测试时代码生成研究","best_for_zh":"设计带沙箱反馈、且不同模型成本差异显著的代码智能体的读者。","confidence":"high","one_line":["PaT tries inexpensive direct code generation first and invokes a planner only after all verified trials fail, improving the code-generation cost-performance frontier.","PaT 先用低成本模型直接生成并执行验证，仅在全部失败时才调用规划器分解问题，从而改善代码生成的成本—性能前沿。"],"why":"It uses an executable failure signal, rather than a learned difficulty score, to decide when an expensive reasoning module is worth calling.","primary_link":"https://aclanthology.org/2026.acl-long.1703/","links":[],"link_count":2,"sections":9},{"id":"personalized-rewardbench-human-aligned-personalization-2026","title":"Personalized RewardBench: Evaluating Reward Models with Human Aligned Personalization","year":2026,"venue":"COLM 2026","authors":["Qiyao Ma","Dechen Gao","Rui Cai","Boqi Zhao","Hanchu Zhou","Junshan Zhang","Zhe Zhao"],"authors_zh":"Qiyao Ma、Dechen Gao、Rui Cai、Boqi Zhao、Hanchu Zhou、Junshan Zhang、Zhe Zhao","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","preference_learning","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["personalization","reward-modeling","alignment"],"tags":["personalization","reward-benchmark","rubrics","colm-2026"],"status":"verified","priority":"可读","paper_type_zh":"个性化奖励模型基准","best_for_zh":"需要评测个体化偏好建模、用户条件化奖励模型或多元价值对齐的研究者","confidence":"high","one_line":["Personalized RewardBench isolates user-specific rubric adherence in chosen–rejected pairs to test whether reward models genuinely model individual preferences.","数据以用户专属量规构造 chosen/rejected 对，隔离个体偏好变量，并验证其可预测 BoN 与 PPO 下游表现。"],"why":"It controls general answer quality so the discriminative signal is the user's own rubric rather than a generic preference proxy.","primary_link":"https://arxiv.org/abs/2604.07343","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Martin-qyma/Personalized-RewardBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/QiyaoMa/Personalized-RewardBench"}],"link_count":4,"sections":9},{"id":"phi-4-reasoning-vision-15b-2026","title":"Phi-4-reasoning-vision-15B Technical Report","year":2026,"venue":"Microsoft Research Technical Report (MSR-TR-2026-10)","authors":["Jyoti Aneja","Michael Harrison","Neel Joshi","Tyler LaBonte","John Langford","Eduardo Salinas"],"authors_zh":"Jyoti Aneja、Michael Harrison、Neel Joshi、Tyler LaBonte、John Langford、Eduardo Salinas","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","agent_training","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["multimodal","mathematics","science","document_understanding","computer_use","safety"],"tags":["phi-4","multimodal-reasoning","frontier-report","data-disclosure-ledger","sft","synthetic-data","computer-use","safety-alignment"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态推理技术报告与数据披露台账","best_for_zh":"需要审查多模态推理模型的数据构造、混合思考模式、反馈边界与可复现性缺口的读者","confidence":"high","one_line":["Phi-4-reasoning-vision-15B reports a 200B-token, three-stage mixed-reasoning SFT pipeline with public source families and data-quality transformations, but withholds training records, internal/acquired data, verification details, and full audit artifacts.","Phi-4-reasoning-vision-15B 披露了约 200B token、三阶段混合推理 SFT、公开源类别与数据质量转换，但未发布训练记录、内部/采购数据、验证细节或完整审计工件。"],"why":"It shows how a frontier report can disclose useful construction interfaces without turning public weights, aggregate counts, or a source list into evidence of reproducible data lineage, feedback, rights, or auditability.","primary_link":"https://www.microsoft.com/en-us/research/publication/phi-4-reasoning-vision-15b-technical-report/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/Phi-4-reasoning-vision-15B"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/microsoft/Phi-4-reasoning-vision-15B"},{"key":"project","label":["Project","项目主页"],"url":"https://www.microsoft.com/en-us/research/blog/phi-4-reasoning-vision-and-the-lessons-of-training-a-multimodal-reasoning-model/"}],"link_count":7,"sections":9},{"id":"deepfakejudge-reasoning-supervision-2026","title":"Pixels Don’t Lie (But Your Detector Might): Bootstrapping MLLM-as-a-Judge for Trustworthy Deepfake Detection and Reasoning Supervision","year":2026,"venue":"CVPR 2026","authors":["Kartik Kuckreja","Parul Gupta","Muhammad Haris Khan","Abhinav Dhall"],"authors_zh":"Kartik Kuckreja、Parul Gupta、Muhammad Haris Khan、Abhinav Dhall","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","pairwise_preference","answer_level"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["trace_writing","release_audit"],"domains":["vision_language","deepfake_detection","reasoning_supervision"],"tags":["multimodal","llm_judge","human_annotation","deepfake"],"status":"verified","priority":"可读","paper_type_zh":"多模态推理监督数据集与 Judge 论文","best_for_zh":"研究多模态推理忠实性、视觉证据标注或高风险 LLM Judge 的研究者。","confidence":"high","one_line":["DeepfakeJudge turns explanation faithfulness in deepfake detection into a human-anchored pointwise and pairwise judge-training dataset.","DeepfakeJudge 以人工视觉证据标注、点式分数和成对偏好，把深伪检测解释的忠实性变为可训练的 Judge 信号。"],"why":"It separates a detector being correct from its explanation being visually grounded and evaluable.","primary_link":"https://openaccess.thecvf.com/content/CVPR2026/html/Kuckreja_Pixels_Dont_Lie_But_Your_Detector_Might_Bootstrapping_MLLM-as-a-Judge_for_CVPR_2026_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/KjAeRsTuIsK/DeepfakeJudge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MBZUAI/DeepfakeJudge-Dataset"}],"link_count":4,"sections":9},{"id":"plan-and-budget-2026","title":"Plan and Budget: Effective and Efficient Test-Time Scaling on Large Language Model Reasoning","year":2026,"venue":"ICLR 2026","authors":["Junhong Lin","Xinyue Zeng","Jie Zhu","Song Wang","Julian Shun","Jun Wu","Dawei Zhou"],"authors_zh":"Junhong Lin、Xinyue Zeng 等（MIT CSAIL、Virginia Tech、弗吉尼亚大学、密歇根州立大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning","mathematics"],"tags":["test-time-compute","budget-allocation","planning","reasoning-efficiency"],"status":"verified","priority":"必读","paper_type_zh":"自适应 token 预算分配与测试时扩展论文（ICLR 2026）","best_for_zh":"设计推理计划、token 预算或审计过度思考的读者。","confidence":"high","one_line":["Plan-and-Budget decomposes a query and adaptively assigns token budgets to subquestions instead of using a fixed thinking length.","Plan-and-Budget 按子问题复杂度分配 token 预算，以避免简单问题过度思考、困难问题思考不足。"],"why":"It makes token allocation an observable decision rather than treating longer reasoning as a free improvement.","primary_link":"https://arxiv.org/abs/2505.16122","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/junhongmit/P-and-B"}],"link_count":4,"sections":9},{"id":"plan-and-budget-2025","title":"Plan and Budget: Effective and Efficient Test-Time Scaling on Reasoning Large Language Models","year":2026,"venue":"ICLR 2026","authors":["Junhong Lin","Xinyue Zeng","Jie Zhu","Song Wang","Julian Shun","Jun Wu","Dawei Zhou"],"authors_zh":"Junhong Lin、Xinyue Zeng、Jie Zhu、Song Wang、Julian Shun、Jun Wu、Dawei Zhou","tracks":["rollout_search_test_time_trace_data"],"source_role":["scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["mathematical_reasoning","instruction_following","travel_planning","test_time_scaling"],"tags":["plan-and-budget","adaptive-token-allocation","test-time-compute","planning","e3","inference-efficiency"],"status":"partial","priority":"可读","paper_type_zh":"测试时计算分配与推理轨迹评估研究","best_for_zh":"需要复现或审计子问题规划、token 分配、任务级 verifier 与测试时计算归因的研究者","confidence":"high","one_line":["Plan-and-Budget uses LLM-produced sub-question plans and credit-weighted schedules to vary test-time compute, with task-specific scoring and E3 aggregation but incomplete released run provenance.","Plan-and-Budget 用 LLM 生成子问题与 credits，并按调度规则分配测试时计算；其 E3 以任务分数和 token 成本汇总，但完整运行轨迹与来源元数据未发布。"],"why":"It makes an allocation policy part of the reasoning record: credits and schedules can affect both output quality and token cost. The release is useful for method inspection and evaluation replication, but absent trace/result manifests and contamination evidence prevent treating it as an auditable training-data release.","primary_link":"https://arxiv.org/abs/2505.16122","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/junhongmit/P-and-B"}],"link_count":4,"sections":9},{"id":"prism-diffusion-tts-2026","title":"Prism: Efficient Test-Time Scaling via Hierarchical Search and Self-Verification for Discrete Diffusion Language Models","year":2026,"venue":"ICML 2026","authors":["Jinbin Bai","Yixuan Li","Yuchen Zhu","Yi Xin","Qingyu Shi","Aosong Feng","Xiaohong Liu","Molei Tao","Jianru Xue","Xiangtai Li","Ming-Hsuan Yang"],"authors_zh":"Jinbin Bai、Yixuan Li、Yuchen Zhu、Yi Xin、Qingyu Shi、Aosong Feng、Xiaohong Liu、Molei Tao、Jianru Xue、Xiangtai Li、Ming-Hsuan Yang（机构：新加坡国立大学、Collov Labs、西安交通大学、佐治亚理工学院等）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","software-engineering"],"tags":["test-time-compute","discrete-diffusion-language-models","search","self-verification","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"测试时计算扩展方法研究（ICML 2026）","best_for_zh":"研究离散扩散语言模型、搜索式推理和验证器替代方案的读者。","confidence":"high","one_line":["Prism allocates denoising-time compute through hierarchical search, partial remasking, and self-verification for discrete diffusion language models.","Prism 以分层搜索、局部重掩码和自验证反馈，为离散扩散语言模型分配更有效的推理计算。"],"why":"It makes the search state, verifier signal, and cost unit explicit for non-autoregressive reasoning inference.","primary_link":"https://arxiv.org/abs/2602.01842","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/viiika/Prism"}],"link_count":4,"sections":9},{"id":"probench-2026","title":"ProBench: Benchmarking GUI Agents with Accurate Process Information","year":2026,"venue":"AAAI Conference on Artificial Intelligence (AAAI-26)","authors":["Leyang Yang","Ziwei Wang","Xiaoxuan Tang","Sheng Zhou","Dajun Chen","Wei Jiang","Yong Li"],"authors_zh":"Leyang Yang, Ziwei Wang, Xiaoxuan Tang, Sheng Zhou, Dajun Chen, Wei Jiang, Yong Li","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","verifier_reward"],"verification_contract":["programmatic","environmental","judgment_required"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","gui_agents","mobile_agents","android","bilingual_evaluation"],"tags":["environment-agent-trajectory-data","gui-agent","mobile-agent","android","interactive-benchmark","process-aware-evaluation","trajectory-level-judge","accessibility-tree","verifier-audit"],"status":"partial","priority":"可读","paper_type_zh":"双语移动GUI智能体过程感知终局评测基准","best_for_zh":"研究移动GUI智能体评测、action evidence、a11y解析与MLLM judge审计的读者","confidence":"medium","one_line":["ProBench evaluates 217 bilingual mobile-GUI tasks by feeding action-level process descriptions and the final screen to a terminal Gemini judge, but releases neither a confirmed executable benchmark package nor a trajectory corpus.","ProBench以a11y converter生成的动作证据和最终截图交给Gemini 2.5 Pro终局judge评测217项双语移动GUI任务，但该证据不是过程监督，且无官方代码、任务、环境或轨迹发布。"],"why":"It demonstrates how a final-state benchmark can detect missed sorting, filtering and other invisible intermediate requirements without prescribing one gold path, while also exposing the audit cost of model-based judging, manual reset, mutable apps and absent record-level releases.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/39974","links":[],"link_count":3,"sections":9},{"id":"probing-preference-representations-mrmbench-2026","title":"Probing Preference Representations: A Multi-Dimensional Evaluation and Analysis Method for Reward Models","year":2026,"venue":"AAAI 2026","authors":["Chenglong Wang","Yifu Huo","Yang Gan","Yongyu Mu","Qiaozhi He","Murun Yang","Bei Li","Chunliang Zhang","Tongran Liu","Anxiang Ma","Zhengtao Yu","Jingbo Zhu","Tong Xiao"],"authors_zh":"Chenglong Wang、Yifu Huo、Yang Gan、Yongyu Mu、Qiaozhi He、Murun Yang、Bei Li、Chunliang Zhang、Tongran Liu、Anxiang Ma、Zhengtao Yu、Jingbo Zhu、Tong Xiao","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reward-modeling","preference-analysis"],"tags":["reward-model","mrmbench","probing","aaai-2026"],"status":"verified","priority":"可读","paper_type_zh":"多维奖励模型基准与表征分析方法","best_for_zh":"需要分析奖励模型偏好维度、可解释性或多目标奖励建模的研究者","confidence":"high","one_line":["MRMBench probes six preference dimensions to diagnose what reward-model representations encode rather than reporting a single aggregate ranking score.","MRMBench 从正确性、帮助性、连贯性等六维剖析偏好表示，提供约 16.7 万实例以诊断奖励模型的真实偏好能力。"],"why":"It turns reward-model evaluation into dimension-level probes and links representation diagnostics to downstream alignment.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40627","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wangclnlp/MRMBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ifnoc/MRMBench"}],"link_count":3,"sections":9},{"id":"prime-process-reinforcement-implicit-rewards-2026","title":"Process Reinforcement through Implicit Rewards","year":2026,"venue":"Transactions on Machine Learning Research 2026","authors":["Ganqu Cui","Lifan Yuan","Zefan Wang","Hanbin Wang","Yuchen Zhang","Jiacheng Chen","Wendi Li","Bingxiang He","Yuchen Fan","Tianyu Yu","Qixin Xu","Weize Chen","Jiarui Yuan","Huayu Chen","Kaiyan Zhang","Xingtai Lv","Shuo Wang","Yuan Yao","Xu Han","Hao Peng","Yu Cheng","Zhiyuan Liu","Maosong Sun","Bowen Zhou","Ning Ding"],"authors_zh":"Ganqu Cui, Lifan Yuan, Zefan Wang, Hanbin Wang, Yuchen Zhang, Jiacheng Chen, Wendi Li, Bingxiang He, Yuchen Fan, Tianyu Yu, Qixin Xu, Weize Chen, Jiarui Yuan, Huayu Chen, Kaiyan Zhang, Xingtai Lv, Shuo Wang, Yuan Yao, Xu Han, Hao Peng, Yu Cheng, Zhiyuan Liu, Maosong Sun, Bowen Zhou, Ning Ding","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","code-reasoning","reinforcement-learning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"在线过程奖励强化学习与偏好数据发布论文","best_for_zh":"需要用响应级正确性标签训练隐式过程奖励模型，或复现数学和代码推理在线强化学习的研究者。","confidence":"high","one_line":["PRIME trains an online implicit process reward model from verifier-derived outcome labels and releases response-level preference data for math and code reasoning.","88,455 指令×8 响应的约 70 万记录，以隐式过程偏好训练 PRM，并用于数学、代码 RL。"],"why":"It makes a concrete connection between inexpensive outcome labels, dense token-level reward estimates, and publicly reusable response records.","primary_link":"https://arxiv.org/abs/2502.01456","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PRIME-RL/PRIME"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PRIME-RL/EurusPRM-Stage1-Data"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/blog/ganqu/prime"}],"link_count":6,"sections":9},{"id":"prm-meet-planning-pddl2prm-2026","title":"Process Reward Models Meet Planning: Generating Precise and Scalable Datasets for Step-Level Rewards","year":2026,"venue":"ACL 2026","authors":["Raffaele Pisano","Roberto Navigli"],"authors_zh":"Pisano and Navigli","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Process Reward Models Meet Planning: Generating Precise and Scalable Datasets for Step-Level Rewards exposes process or trace supervision data.","提出 PDDL2PRM：把可执行规划中的动作转为带精确分级奖励的推理步骤，自动生成约百万规模 PRM 训练数据。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2604.17957","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Babelscape/PDDL2PRM"},{"key":"project","label":["Project","项目主页"],"url":"https://babelscape.github.io/prm-meets-planning/"}],"link_count":3,"sections":9},{"id":"profit-token-selection-sft-2026","title":"ProFit: Leveraging High-Value Signals in SFT via Probability-Guided Token Selection","year":2026,"venue":"Findings of ACL 2026","authors":["Tao Liu","Taiqiang Wu","Runming Yang","Shaoning Sun","Junjie Wang","Yujiu Yang"],"authors_zh":"Tao Liu, Taiqiang Wu, Runming Yang, Shaoning Sun, Junjie Wang, Yujiu Yang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["step_level"],"training_use":["sft","rlvr"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["instruction-tuning","reasoning","mathematics"],"tags":["token-selection","sft","probability","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"概率引导的词元级监督微调研究","best_for_zh":"设计推理导向监督微调词元选择或掩码的读者。","confidence":"high","one_line":["ProFit masks low-probability reference tokens during SFT so high-probability semantic anchors dominate the training signal.","ProFit 在监督微调中屏蔽低概率参考词元，使高概率语义锚点主导训练信号。"],"why":"It makes the token-level boundary of the SFT supervision record explicit.","primary_link":"https://aclanthology.org/2026.findings-acl.755/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Utaotao/ProFit"}],"link_count":3,"sections":9},{"id":"icpo-self-improvement-2026","title":"Provable and Practical In-Context Policy Optimization for Self-Improvement","year":2026,"venue":"ICLR 2026","authors":["Tianrun Yu","Yuxiao Yang","Zhaoyang Wang","Kaixiang Zhao","Porter Jenkins","Xuchao Zhang","Chetan Bansal","Huaxiu Yao","Weitong Zhang"],"authors_zh":"Tianrun Yu、Yuxiao Yang、Zhaoyang Wang、Kaixiang Zhao、Porter Jenkins、Xuchao Zhang、Chetan Bansal、Huaxiu Yao、Weitong Zhang","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","scaling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配的读者。","confidence":"high","one_line":["Provable and Practical In-Context Policy Optimization for Self-Improvement","自我反思可能消耗许多轮次，却不知道自评奖励是否可靠。"],"why":"It measures a concrete decision about additional inference computation.","primary_link":"https://arxiv.org/abs/2603.01335","links":[],"link_count":3,"sections":9},{"id":"pythagoras-sft-distillation-2026","title":"Pythagoras-Prover: Advancing Efficient Formal Proving via Augmented Lean Formalisation","year":2026,"venue":"arXiv preprint","authors":["Leang, Joshua Ong Jun","Zhao, Zheng","Stoian, Mihaela Catalina","Xu, Qiyuan","Li, Haonan","Li, Wenda","Cohen, Shay B.","Giunchiglia, Eleonora"],"authors_zh":"Leang, Joshua Ong Jun、Zhao, Zheng、Stoian, Mihaela Catalina、Xu, Qiyuan、Li, Haonan、Li, Wenda、Cohen, Shay B.、Giunchiglia, Eleonora","tracks":["instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["compiler-verified-formal-proof-distillation"],"tags":["instruction-demonstration-rationale","arxiv-2606.12594","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"自回归与扩散式定理证明器监督微调","confidence":"high","one_line":["Augmented Lean Formalisation mutates seed problems, writes reasoning plans and Lean artifacts, and retains only compiler-verified proofs in a 336K-record SFT set.","Pythagoras 把题目变异、证明计划和 Lean 证明结合，公开 336596 条经编译器验证的形式证明监督记录。"],"why":"Formal proving data is limited by the cost of translating informal mathematics into diverse, executable Lean statements and proofs.","primary_link":"https://arxiv.org/abs/2606.12594","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Pythagoras-LM/SFT_Dataset_Distillation_4B"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/Pythagoras-LM"}],"link_count":3,"sections":9},{"id":"qed-nano-2026","title":"QED-Nano: Teaching a Tiny Model to Prove Hard Theorems","year":2026,"venue":"arXiv preprint","authors":["LM-Provers","Yuxiao Qu","Amrith Setlur","Jasper Dekoninck","Edward Beeching","Jia Li","Ian Wu","Lewis Tunstall","Aviral Kumar"],"authors_zh":"LM-Provers、Yuxiao Qu、Amrith Setlur、Jasper Dekoninck、Edward Beeching、Jia Li、Ian Wu、Lewis Tunstall、Aviral Kumar","tracks":["rollout_search_test_time_trace_data","preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["model_report","data_release","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematics","olympiad_mathematics","natural_language_theorem_proving","long_horizon_reasoning","test_time_reasoning"],"tags":["qed-nano","fineproofs-rl","olympiad-proof","natural-language-theorem-proving","rubric-reward","llm-judge","grpo","reasoning-cache","rollout-population","curriculum-filtering","test-time-scaling","missing-raw-rollouts"],"status":"partial","priority":"必读","paper_type_zh":"自然语言定理证明模型报告、rubric 奖励配方与统计型 rollout 数据发布","best_for_zh":"研究 rollout 数据、Olympiad proof RL、LLM judge 奖励、Reasoning Cache、test-time compute 与数据发布可审计性的读者","confidence":"high","one_line":["QED-Nano couples 128-attempt difficulty statistics, 16-rollout rubric-reward GRPO, and three-turn Reasoning-Cache training, while FineProofs-RL releases prompts, rubrics, and score arrays but not the proofs behind them.","QED-Nano 将每题名义 128 次的离线难度估计、每题 16 次的 rubric-reward GRPO 与三轮 Reasoning Cache 训练连接起来；FineProofs-RL 发布题目、rubric 和分数数组，但不发布产生这些分数的证明文本或逐次评分意见。"],"why":"The work makes rollout budgeting, curriculum selection, judgment-based proof rewards, and train/test-time scaffold coupling explicit for a small open model. It also demonstrates that score populations can support prompt selection and RL without constituting a reusable trace corpus.","primary_link":"https://arxiv.org/abs/2604.04898","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/CMU-AIRe/QED-Nano"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lm-provers/FineProofs-RL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/lm-provers/qed-nano"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/lm-provers/qed-nano-blogpost"}],"link_count":7,"sections":9},{"id":"quantifying-biases-judge-evals-2026","title":"Quantifying biases in LLM-as-Judge evals","year":2026,"venue":"ICML 2026","authors":["Magda Dubois","Harry Coppock","Mario Giulianelli","Ole Kristian Jorgensen","Timo Flesch","Lennart Luettgau","Cozmin Ududec"],"authors_zh":"Magda Dubois、Harry Coppock、Mario Giulianelli、Ole Kristian Jorgensen、Timo Flesch、Lennart Luettgau、Cozmin Ududec","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"LLM 自动评测偏差的贝叶斯建模与校正研究","best_for_zh":"需要量化 verbosity、自偏好等评测偏差并报告不确定性的研究者。","confidence":"medium","one_line":["Models, quantifies, and corrects systematic biases in LLM autograders.","以贝叶斯广义线性模型识别、量化并校正 LLM 自动评测器的系统偏差。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=eQxVeNZcYT","links":[],"link_count":1,"sections":9},{"id":"qwen-agentworld-2026","title":"Qwen-AgentWorld: Language World Models for General Agents","year":2026,"venue":"arXiv preprint","authors":["Yuxin Zuo","Zikai Xiao","Li Sheng","Fei Huang","Jianhong Tu","Yuxuan Liu","Tianyi Tang","Xiaomeng Hu","Yang Su","Qingfeng Lan","Yantao Liu","Qin Zhu","Yinger Zhang","Bowen Yu","Haiquan Zhao","Haiyang Xu","Jianxin Yang","Jiayang Cheng","Junyang Wang","Lianghao Deng","Mingfeng Xue","Tianyi Bai","Yang Fan","Yubo Ma","Yucheng Li","Zeyu Cui","Zhihai Wang","Zhihui Xie","Zhuorui Ye","An Yang","Dayiheng Liu","Jingren Zhou","Ning Ding"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","agent_environment","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["agentic-environments","tool-use","software-engineering","web","operating-systems"],"tags":["qwen","frontier-report","data-disclosure-ledger","language-world-model","agent-environment","agentworldbench","cpt","sft","rl","gspo","rubric-reward","rule-verifier"],"status":"partial","priority":"必读","paper_type_zh":"智能体环境语言世界模型报告与数据披露账本","best_for_zh":"需要审计智能体轨迹、世界模型训练、LLM 裁判奖励和评测数据边界的读者","confidence":"medium","one_line":["Qwen-AgentWorld reports CPT, SFT, and GSPO RL over more than 10M seven-domain environment trajectories, releases 35B weights and the separate 2,170-sample AgentWorldBench test set, and leaves the training corpus and reward stack only partially auditable.","Qwen-AgentWorld 报告在七类环境的 1,000 多万条轨迹上依次进行 CPT、SFT 和 GSPO RL，并使用 rubric 加规则的混合奖励；其发布了 35B 权重和独立的 AgentWorldBench 评测集，但训练数据、奖励与环境复现仍不完整。"],"why":"It exposes enough of a frontier agent-world-model pipeline to distinguish trajectory sources, next-state targets, SFT curation, and RL feedback, while showing why an open benchmark and checkpoint do not make the underlying training data reproducible.","primary_link":"https://arxiv.org/abs/2606.24597","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen-AgentWorld"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Qwen/AgentWorldBench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen-AgentWorld-35B-A3B"},{"key":"project","label":["Project","项目主页"],"url":"https://qwen.ai/blog?id=qwen-agentworld"}],"link_count":6,"sections":9},{"id":"qwen3-coder-next-technical-report-2026","title":"Qwen3-Coder-Next Technical Report","year":2026,"venue":"arXiv preprint","authors":["Ruisheng Cao","Mouxiang Chen","Jiawei Chen","Zeyu Cui","Yunlong Feng","Binyuan Hui","Yuheng Jing","Kaixin Li","Mingze Li","Junyang Lin","Zeyao Ma","Kashun Shum","Xuwu Wang","Jinxi Wei","Jiaxi Yang","Jiajun Zhang","Lei Zhang","Zongmeng Zhang","Wenting Zhao","Fan Zhou"],"authors_zh":"Ruisheng Cao、Mouxiang Chen、Jiawei Chen、Zeyu Cui、Yunlong Feng、Binyuan Hui、Yuheng Jing、Kaixin Li、Mingze Li、Junyang Lin、Zeyao Ma、Kashun Shum、Xuwu Wang、Jinxi Wei、Jiaxi Yang、Jiajun Zhang、Lei Zhang、Zongmeng Zhang、Wenting Zhao、Fan Zhou","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","distillation","preference_learning","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["software_engineering","coding","agentic_tool_use","web"],"tags":["qwen3-coder-next","frontier-report","data-disclosure-ledger","coding-agent","executable-environment","github-pr","agentic-training","rlvr","reward-hacking"],"status":"partial","priority":"必读","paper_type_zh":"前沿编码智能体技术报告与数据披露账本","best_for_zh":"关注可执行软件工程任务、agentic RL 反馈、reward-hacking 控制和复现审计的研究者","confidence":"high","one_line":["Qwen3-Coder-Next discloses executable coding-task synthesis and Docker-based verification with staged agentic training, but not its training tasks, environments, trajectories, or audit records.","Qwen3-Coder-Next 披露了可执行编码任务合成、Docker 验证和分阶段 agentic training，但未发布训练任务、环境、轨迹或审计记录。"],"why":"It separates released weights and inference code from partially disclosed task provenance, verification, data rights, and reproducibility evidence.","primary_link":"https://arxiv.org/abs/2603.00729","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen3-Coder"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen3-Coder-Next"}],"link_count":4,"sections":9},{"id":"qwen3-5-omni-2026","title":"Qwen3.5-Omni Technical Report","year":2026,"venue":"arXiv preprint","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","distillation","preference_learning","rlvr","agent_training"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["multimodal","audio"],"tags":["frontier-report","data-disclosure-ledger","qwen","omnimodal","audio-visual","aria"],"status":"partial","priority":"暂缓","paper_type_zh":"前沿全模态模型技术报告与数据披露台账","best_for_zh":"审计全模态语料声明与缺失后训练证据之间差距的读者。","confidence":"low","one_line":["Qwen3.5-Omni reports aggregate text-vision and 100M-plus-hour audio-visual training scale plus ARIA, but its official abstract leaves post-training data, feedback, and audit artifacts undisclosed.","Qwen3.5-Omni 报告文本—视觉和超过一亿小时音视频语料、256K 上下文与全模态能力，但未披露后训练数据、奖励/verifier、rollout 或审计细节。"],"why":"The entry records the meaningful aggregate disclosure without treating it as a reproducible dataset, reward system, or reasoning-training recipe.","primary_link":"https://arxiv.org/abs/2604.15804","links":[],"link_count":2,"sections":9},{"id":"qwen3-5-2026","title":"Qwen3.5: Towards Native Multimodal Agents","year":2026,"venue":"Qwen official release blog","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["agent_training"],"construction_layer":["optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["multimodal","visual_reasoning","reasoning","agentic_tool_use"],"tags":["qwen3-5","qwen","frontier-report","data-disclosure-ledger","multimodal-agent","agent-training","asynchronous-rl","open-weights"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态智能体发布报告与数据披露台账","best_for_zh":"需要审计多模态智能体 RL、环境披露与开放权重边界的读者","confidence":"medium","one_line":["Qwen3.5 reports scaled multimodal and multi-turn RL with agent-oriented environments and releases a post-trained checkpoint, but not the tasks, traces, rewards, environment artifacts, or provenance needed to audit or reproduce that training.","Qwen3.5 报告了在面向智能体环境中扩展多模态和多轮 RL，并发布后训练检查点；但未发布审计或复现该训练所需的任务、轨迹、奖励、环境工件或溯源信息。"],"why":"It is a clear disclosure-boundary case: systems-level RL scaling and released weights provide useful ledger evidence, but cannot be treated as an open agent-data, reward, or reproducible RL recipe release.","primary_link":"https://qwen.ai/blog?id=qwen3.5","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B"}],"link_count":3,"sections":9},{"id":"r-diverse-self-play-2026","title":"R-Diverse: Mitigating Diversity Illusion in Self-Play LLM Training","year":2026,"venue":"ICML 2026 regular","authors":["Gengsheng Li","Jinghan He","Shijie Wang","Ruiqi Liu","Renrui Zhang","Zijun Yao","Junfeng Fang","Haiyun Guo","Dan Zhang","Jinqiao Wang"],"authors_zh":"Gengsheng Li、Jinghan He、Shijie Wang、Ruiqi Liu、Renrui Zhang、Zijun Yao、Junfeng Fang、Haiyun Guo、Dan Zhang、Jinqiao Wang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training"],"construction_layer":["prompt_sourcing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematical_reasoning","general_reasoning"],"tags":["self-play","persistent-memory","diversity-illusion","pseudo-labeling","grpo","release-gap"],"status":"partial","priority":"必读","paper_type_zh":"记忆增强推理自博弈方法","best_for_zh":"关注自博弈课程、多样性度量、经验回放与数据审计的研究者","confidence":"medium","one_line":["Adds skill-aware historical diversity penalties and replay to Challenger-Solver self-play, but releases no executable pipeline or generated records.","为 Challenger-Solver 自博弈加入程序级技能多样性和历史记忆惩罚，但未发布可执行流程或生成记录。"],"why":"Shows how within-batch lexical novelty can hide repeated reasoning procedures and how persistent memory changes both the generated curriculum and its audit surface.","primary_link":"https://openreview.net/forum?id=DZiuKVvrJW","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/Gengsheng-Li/R-Diverse"}],"link_count":5,"sections":9},{"id":"ranking-reasoning-tts-2026","title":"Ranking Reasoning LLMs under Test-Time Scaling","year":2026,"venue":"ACL 2026","authors":["Mohsen Hariri","Michael Hinczewski","Jing Ma","Vipin Chaudhary"],"authors_zh":"Mohsen Hariri、Michael Hinczewski、Jing Ma、Vipin Chaudhary（机构：Case Western Reserve University）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","evaluation","ranking","repeated-sampling","mathematical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"测试时扩展评测与统计排序研究（ACL 2026）","best_for_zh":"在重复推理条件下评测或选择推理模型的读者。","confidence":"high","one_line":["Scorio evaluates how statistical ranking remains stable when reasoning models receive different test-time sampling budgets.","Scorio 研究推理模型在不同推理时采样预算下如何获得稳定、可信的排名。"],"why":"It treats the number of test-time trials as part of the evaluation protocol, not a hidden implementation detail.","primary_link":"https://aclanthology.org/2026.acl-long.1544/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mohsenhariri/scorio"}],"link_count":4,"sections":9},{"id":"random-calculation-contamination-2025","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","year":2026,"venue":"AAAI 2026","authors":["Mingqi Wu","Zhihao Zhang","Qiaole Dong","Zhiheng Xi","Jun Zhao","Senjie Jin","Xiaoran Fan","Yuhao Zhou","Huijie Lv","Ming Zhang","Yanwei Fu","Qin Liu","Songyang Zhang","Qi Zhang"],"authors_zh":"Mingqi Wu、Zhihao Zhang、Qiaole Dong、Zhiheng Xi、Jun Zhao、Senjie Jin、Xiaoran Fan、Yuhao Zhou、Huijie Lv、Ming Zhang、Yanwei Fu、Qin Liu、Songyang Zhang、Qi Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["mathematical_reasoning","arithmetic"],"tags":["randomcalculation","procedural-data","contamination","rlvr","grpo","structural-memorization"],"status":"partial","priority":"必读","paper_type_zh":"程序化算术数据发布与污染审计","best_for_zh":"关注RLVR、程序化验证、基准污染和记忆诊断的研究者","confidence":"high","one_line":["RandomCalculation releases 20 levels of procedurally generated arithmetic prompt-answer pairs and uses two levels to test whether RLVR gains survive a lower-contamination control.","程序化生成20个算术难度层级，并以正确/随机/反向奖励对照区分低污染RLVR学习与基准记忆。"],"why":"The work joins a reusable construction recipe, a programmatic answer contract, reward-control ablations, and explicit contamination risks, while leaving seeds, exact splits, structural decontamination, and release licensing unresolved.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/download/40687/44648","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wumingqi/LLM-Math-Evaluation"},{"key":"data","label":["Data","数据"],"url":"https://github.com/wumingqi/LLM-Math-Evaluation/tree/main/random_calculation/result"}],"link_count":6,"sections":9},{"id":"omnithought-2026","title":"Reasoning with OmniThought: A Large CoT Dataset with Verbosity and Cognitive Difficulty Annotations","year":2026,"venue":"ACL 2026","authors":["Wenrui Cai","Chengyu Wang","Junbing Yan","Jun Huang","Xiangzhong Fang"],"authors_zh":"Wenrui Cai、Chengyu Wang、Junbing Yan、Jun Huang、Xiangzhong Fang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","distillation","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["math","code","science","reasoning"],"tags":["omnithought","chain-of-thought","synthetic-reasoning-data","multi-teacher-distillation","reasoning-verbosity","cognitive-difficulty","llm-as-a-judge","metadata-selection","math","code","science"],"status":"partial","priority":"必读","paper_type_zh":"多教师 CoT 数据发布、验证与元数据条件选择配方","best_for_zh":"构建或审计推理蒸馏数据、按模型能力选择轨迹、设计偏好对或 RV/CD 奖励的研究者","confidence":"medium","one_line":["OmniThought releases nested multi-teacher math, code, and science traces with separate reasoning-validity and answer acceptance signals plus RV/CD annotations, while row-level prompt provenance, generation settings, executable verifier assets, and exact nested-trace lineage remain incomplete.","OmniThought 发布 708,009 个问题行及其嵌套的多教师数学、代码与科学 CoT，分别保留推理有效性、答案接收信号和 RV/CD 标注；但逐行题源、生成设置、可执行验证器资产与精确轨迹 lineage 仍不完整。"],"why":"It turns trace length, judged cognitive difficulty, teacher identity, and final-answer validity into inspectable construction variables for SFT, preference data, and RL rewards, making it a useful open-recipe case with clear audit boundaries.","primary_link":"https://aclanthology.org/2026.acl-long.382/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/modelscope/easydistill"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/alibaba-pai/OmniThought"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/alibaba-pai/distilqwen"}],"link_count":7,"sections":9},{"id":"reasoningbank-matts-2026","title":"ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","year":2026,"venue":"ICLR 2026","authors":["Siru Ouyang","Jun Yan","I-Hung Hsu","Yanfei Chen","Ke Jiang","Zifeng Wang","Rujun Han","Long T. Le","Samira Daruki","Xiangru Tang","Vishy Tirumalashetty","George Lee","Mahsan Rofouei","Hangfei Lin","Jiawei Han","Chen-Yu Lee","Tomas Pfister"],"authors_zh":"Siru Ouyang、Jun Yan、I-Hung Hsu、Yanfei Chen、Ke Jiang、Zifeng Wang、Rujun Han、Long T. Le、Samira Daruki、Xiangru Tang、Vishy Tirumalashetty、George Lee、Mahsan Rofouei、Hangfei Lin、Jiawei Han、Chen-Yu Lee、Tomas Pfister（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","持续运行的代理会反复丢弃成功和失败交互中的洞见，因此更多测试时探索可能重复同样错误，而不是积累有用经验。"],"why":"It makes an inference-budget decision auditable.","primary_link":"https://arxiv.org/abs/2509.25140","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/google-research/reasoning-bank"},{"key":"project","label":["Project","项目主页"],"url":"https://research.google/blog/reasoningbank-enabling-agents-to-learn-from-experience/"}],"link_count":5,"sections":9},{"id":"refact-scientific-confabulation-2026","title":"ReFACT: A Benchmark for Scientific Confabulation Detection with Positional Error Annotations","year":2026,"venue":"EACL 2026","authors":["Yindong Wang","Martin Preiß","Margarita Bugueño","Jan Vincent Hoffbauer","Abdullatif Ghajar","Tolga Buz","Gerard de Melo"],"authors_zh":"Yindong Wang、Martin Preiß、Margarita Bugueño、Jan Vincent Hoffbauer、Abdullatif Ghajar、Tolga Buz、Gerard de Melo","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["scientific-factuality","error-localization","process-evaluation"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究科学事实性、细粒度错误定位和 LLM 评审可靠性的读者。","confidence":"high","one_line":["ReFACT pairs scientific answers with expert-marked error spans to test detection, localization, and correction of confabulations.","ReFACT 以专家标注的科学错误位置配对答案，评测幻觉检测、定位与纠正。"],"why":"It replaces binary factuality labels with reusable positional error supervision for scientific-answer verification.","primary_link":"https://aclanthology.org/2026.eacl-long.381/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ddz5431/ReFACT"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ddz5431/refact"}],"link_count":5,"sections":9},{"id":"proofbench-fine-grained-math-proofs-2026","title":"Reliable Fine-Grained Evaluation of Natural Language Math Proofs","year":2026,"venue":"ICLR 2026","authors":["Wenjie Ma","Andrei Cojocaru","Neel Kolhe","Haihan Zhang","Vincent Zhuang","Matei Zaharia","Sewon Min"],"authors_zh":"Wenjie Ma、Andrei Cojocaru、Neel Kolhe、Haihan Zhang、Vincent Zhuang、Matei Zaharia、Sewon Min","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"必读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["ProofBench makes local errors in natural-language mathematical proofs visible through reliable fine-grained evaluation.","以细粒度判定把自然语言数学证明的局部错误定位转化为可靠评测。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://openreview.net/pdf?id=ky5iqwZSXI","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/wenjiema02/ProofBench"}],"link_count":2,"sections":9},{"id":"reprobe-internal-state-tts-2026","title":"ReProbe: Efficient Test-Time Scaling of Multi-Step Reasoning by Probing Internal States of Large Language Models","year":2026,"venue":"ACL 2026","authors":["Jingwei Ni","Ekaterina Fadeeva","Tianyi Wu","Mubashara Akhtar","Jiaheng Zhang","Elliott Ash","Markus Leippold","Timothy Baldwin","See-Kiong Ng","Artem Shelmanov","Mrinmaya Sachan"],"authors_zh":"Jingwei Ni、Ekaterina Fadeeva、Tianyi Wu 等（机构：ETH Zürich、National University of Singapore、MBZUAI、University of Zürich、The University of Melbourne）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","planning","general-question-answering"],"tags":["test-time-compute","process-verifier","internal-states","beam-search","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"步骤级验证与高效测试时扩展研究（ACL 2026）","best_for_zh":"希望降低最佳候选或束搜索推理中验证器成本的读者。","confidence":"high","one_line":["ReProbe uses a small probe over a frozen LLM’s internal states to verify reasoning steps during test-time scaling.","ReProbe 用冻结语言模型的内部状态训练小型探针，在测试时扩展中验证推理步骤。"],"why":"It makes step verification depend on signals already produced by the generator, reducing the cost of allocating inference computation.","primary_link":"https://aclanthology.org/2026.acl-long.536/","links":[{"key":"code","label":["Code","代码"],"url":"https://reprobe.github.io/"}],"link_count":3,"sections":9},{"id":"cherrl-2026","title":"Reproducing, Analyzing, and Detecting Reward Hacking in Rubric-Based Reinforcement Learning","year":2026,"venue":"arXiv","authors":["Xuekang Wang","Zhuoyuan Hao","Shuo Hou","Hao Peng","Juanzi Li","Xiaozhi Wang"],"authors_zh":"Xuekang Wang, Zhuoyuan Hao, Shuo Hou, Hao Peng, Juanzi Li, Xiaozhi Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","reward-hacking"],"status":"verified","priority":"可读","paper_type_zh":"奖励投机审计与检测环境论文","best_for_zh":"需要审计量规奖励、LLM judge 偏差或训练轨迹异常的研究者。","confidence":"high","one_line":["controllable LLM-judge bias and reward-hacking environment","CHERRL 通过注入可控的 LLM judge 偏差，复现并定位基于量规强化学习中的奖励投机。"],"why":"It offers a concrete audit surface for reward-hacking failure modes.","primary_link":"https://arxiv.org/abs/2606.04923","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUAIS-Lab/CHERRL"}],"link_count":2,"sections":9},{"id":"researchclawbench-autonomous-research-2026","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","year":2026,"venue":"arXiv","authors":["Wanghan Xu","Yuhao Zhou","Yifan Zhou","Qinglong Cao","Shuo Li","Jia Bu","Bo Liu","Yixin Chen","Xuming He","Xiangyu Zhao"],"authors_zh":"Wanghan Xu, Yuhao Zhou, Yifan Zhou, Qinglong Cao, Shuo Li 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","agent_environment","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["full_episode"],"training_use":["evaluation","reward_modeling","agent_training","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["science","agent_workflows"],"tags":["science","agents","rubric","discovery"],"status":"verified","priority":"可读","paper_type_zh":"端到端自主科学研究的环境与 rubric 基准","best_for_zh":"需要评估科研 Agent、实验轨迹或科学产物 Judge 的研究者。","confidence":"high","one_line":["ResearchClawBench grades end-to-end scientific re-discovery from raw data using hidden-paper, expert-weighted artifacts.","ResearchClawBench 以隐藏目标论文和专家加权产物准则，衡量 AI 的端到端科学复现与发现能力。"],"why":"It evaluates a scientific workflow rather than isolated question answering or code generation.","primary_link":"https://arxiv.org/abs/2606.07591","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InternScience/ResearchClawBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/InternScience/ResearchClawBench"},{"key":"project","label":["Project","项目主页"],"url":"https://internscience.github.io/ResearchClawBench-Home/"}],"link_count":5,"sections":9},{"id":"researchqa-scholarly-qa-2025","title":"ResearchQA: Evaluating Scholarly Question Answering at Scale Across 75 Fields with Survey-Mined Questions and Rubrics","year":2026,"venue":"TACL","authors":["Li S. Yifei","Allen Chang","Chaitanya Malaviya","Mark Yatskar"],"authors_zh":"Li S. Yifei, Allen Chang, Chaitanya Malaviya, Mark Yatskar","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["science","scholarly_qa","cross_domain"],"tags":["scholarly_qa","rubric","science"],"status":"verified","priority":"必读","paper_type_zh":"多学科学术问答的 survey-mined rubric 基准","best_for_zh":"研究科研问答、引用评测与跨领域 RAG 的读者。","confidence":"high","one_line":["ResearchQA turns surveys from 75 fields into 21K scholarly queries and 160K evaluation rubrics.","ResearchQA 从 75 个领域的综述文章提炼问题与细粒度 rubric，评测学术问答的引用、解释与局限性。"],"why":"It exposes citation, limitation, and comparison gaps in scholarly answers across disciplines.","primary_link":"https://arxiv.org/abs/2509.00496","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/realliyifei/ResearchQA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/realliyifei/ResearchQA"},{"key":"project","label":["Project","项目主页"],"url":"https://researchqa.cylumn.com/"}],"link_count":5,"sections":9},{"id":"reasoning-benchmark-comparability-survey-2026","title":"Rethinking Benchmark Comparability: A Survey of Reasoning Benchmarks for Large Language Models","year":2026,"venue":"Preprints.org","authors":["Chenyuan Zhang","Simin Liu","Hanjing Li","Te Gao","Yidi Wang","Qiguang Chen","Xiachong Feng","Li Cai","Mengnan Du","Zhuotao Tian","Libo Qin","Philip S. Yu","Min Zhang"],"authors_zh":"Chenyuan Zhang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["reasoning","reasoning-data","benchmarks","evaluation","data-quality"],"tags":["foundations-and-primers","reasoning-benchmarks","evaluation","survey",2026],"status":"verified","priority":"必读","paper_type_zh":"推理基准综述","best_for_zh":"选择、构建或解读推理基准与数据的读者。","confidence":"high","one_line":["A 2026 survey that explains why two “reasoning benchmark” scores are often not measuring the same thing.","解释推理基准分数为何常常不可直接比较的 2026 综述。"],"why":"It turns benchmark selection from leaderboard following into a concrete comparison of capability, conditions, and metric.","primary_link":"https://www.preprints.org/manuscript/202605.0806","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/chenyuanTKCY/Awesome-Benchmarks-for-LLM-Reasoning"}],"link_count":3,"sections":9},{"id":"representation-as-a-judge-2026","title":"Rethinking LLM-as-a-Judge: Representation-as-a-Judge with Small Language Models via Semantic Capacity Asymmetry","year":2026,"venue":"ICLR 2026","authors":["Zhuochun Li","Yong Zhang","Ming Li","Yuelyu Ji","Yiming Zeng","Ning Cheng","Yun Zhu","Yanmeng Wang","Shaojun Wang","Jing Xiao","Daqing He"],"authors_zh":"Zhuochun Li, Yong Zhang, Ming Li, Yuelyu Ji, Yiming Zeng, Ning Cheng, Yun Zhu, Yanmeng Wang, Shaojun Wang, Jing Xiao, Daqing He","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"表示探测式 LLM judge 与推理数据过滤研究","best_for_zh":"需要低成本、可解释地评估推理回答或过滤噪声轨迹的研究者。","confidence":"high","one_line":["open judge data and reproducible representation-based evaluator","INSPECTOR 以小语言模型的隐藏表示预测评测分数，避免依赖生成式 LLM judge。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://openreview.net/forum?id=VAISvCsrvG","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhuochunli/Representation-as-a-judge"}],"link_count":3,"sections":9},{"id":"dysql-bench-dynamic-multiturn-text-to-sql-2026","title":"Rethinking Text-to-SQL: Dynamic Multi-turn SQL Interaction for Real-world Database Exploration","year":2026,"venue":"Findings of ACL 2026","authors":["Linzhuang Sun","Tianyu Guo","Hao Liang","Ruitong Liu","Yuying Li","Qifeng Cai","Jingxuan Wei","Yuchen Wu","Bihui Yu","Xiangxiang Zhang","Wentao Zhang","Bin Cui"],"authors_zh":"Linzhuang Sun, Tianyu Guo, Hao Liang, Ruitong Liu, Yuying Li, Qifeng Cai, Jingxuan Wei, Yuchen Wu, Bihui Yu, Xiangxiang Zhang, Wentao Zhang, Bin Cui","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code","database","reasoning"],"tags":["text-to-sql","database","multiturn","execution","acl-2026"],"status":"verified","priority":"可读","paper_type_zh":"可执行的多轮 Text-to-SQL 交互基准","best_for_zh":"研究数据库代理、多轮 SQL 交互或执行式奖励的研究者。","confidence":"high","one_line":["DySQL-Bench evaluates multi-turn SQL agents by executing their CRUD actions on database snapshots and checking both query results and state changes.","DySQL-Bench 在数据库快照中执行多轮 SQL 的 CRUD 操作，并同时检查查询结果和状态变化来评估代理。"],"why":"It moves Text-to-SQL evaluation from static query matching to stateful, executable database interaction.","primary_link":"https://aclanthology.org/2026.findings-acl.1654/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Aurora-slz/DySQL-Bench"}],"link_count":4,"sections":9},{"id":"retraceqa-commonsense-reasoning-traces-2026","title":"ReTraceQA: Evaluating Reasoning Traces of Small Language Models in Commonsense Question Answering","year":2026,"venue":"ACL 2026","authors":["Francesco Maria Molfese","Luca Moroni","Ciro Porcaro","Simone Conia","Roberto Navigli"],"authors_zh":"Francesco Maria Molfese, Luca Moroni, Ciro Porcaro, Simone Conia, Roberto Navigli","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["commonsense-reasoning","process-reward-modeling","evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要审查小模型常识推理过程而非仅比较最终答案的研究者。","confidence":"high","one_line":["ReTraceQA adds expert first-error and error-category annotations to 2,421 commonsense reasoning traces.","ReTraceQA 为 2,421 条常识推理轨迹提供专家首错与错误类别标注，揭示答案正确但过程错误的失真。"],"why":"The release provides a reusable process-supervision or process-evaluation surface.","primary_link":"https://aclanthology.org/2026.acl-long.1798/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sapienzanlp/ReTraceQA"}],"link_count":3,"sections":9},{"id":"image-grounded-cot-survey-2026","title":"Revealing the Seen, Imagining the Beyond: A Survey of Image-Grounded Chain-of-Thought Reasoning in Multimodal LLMs","year":2026,"venue":"ACL 2026 Long Papers","authors":["Qihua Dong","Yitian Zhang","Huimin Zeng","Yizhou Wang","Jianglin Lu","Kuo Yang","Yun Fu"],"authors_zh":"Qihua Dong 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","multimodal","chain-of-thought","reasoning-data"],"tags":["foundations-and-primers","multimodal","chain-of-thought","acl-2026","survey"],"status":"verified","priority":"必读","paper_type_zh":"图像扎根思维链综述","best_for_zh":"构建或评估多模态图像扎根推理数据的读者。","confidence":"high","one_line":["An ACL survey of multimodal reasoning processes that alternate between text and visual state updates.","梳理文字理由与视觉状态交替更新的多模态推理过程的 ACL 综述。"],"why":"It makes the intermediate visual state a first-class object when judging whether a multimodal explanation is grounded.","primary_link":"https://aclanthology.org/2026.acl-long.2087/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/dddraxxx/Awesome-Image-Grounded-CoT"}],"link_count":3,"sections":9},{"id":"selfplay-prompt-difficulty-2026","title":"Revisiting Self-Play Preference Optimization: On the Role of Prompt Difficulty","year":2026,"venue":"Findings of ACL 2026","authors":["Yao Xiao","Jung-jae Kim","Roy Ka-wei Lee","Lidong Bing"],"authors_zh":"Yao Xiao, Jung-jae Kim, Roy Ka-wei Lee, Lidong Bing","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["self-play","preference-optimization","prompt-selection","difficulty"],"status":"verified","priority":"必读","paper_type_zh":"自博弈偏好优化的提示难度选择研究","best_for_zh":"适合在构造偏好对前，需要先为自博弈 DPO 流程过滤提示、节省轨迹与奖励模型预算的读者。","confidence":"high","one_line":["The paper improves self-play DPO by ranking prompts with mean reward and retaining only the easiest 30 percent rather than mixing in difficult prompts.","该论文以平均奖励排序提示，仅保留最容易的 30% 来构造自博弈 DPO 偏好对。"],"why":"It identifies the prompt as a first-class consumer-side decision in preference-data construction, not just a container for response sampling.","primary_link":"https://aclanthology.org/2026.findings-acl.144/","links":[],"link_count":2,"sections":9},{"id":"reward-hacking-benchmark-2025","title":"Reward Hacking Benchmark: Measuring Exploits in LLM Agents with Tool Use","year":2026,"venue":"ICML 2026 (accepted)","authors":["Kunvar Thaman"],"authors_zh":"Kunvar Thaman","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"奖励投机与评测可靠性基准","best_for_zh":"需要审计智能体奖励投机风险的研究者。","confidence":"medium","one_line":["Public release of deliberately vulnerable code-optimization environments for measuring reward hacking.","RHB以多步工具任务和自然捷径机会，测量语言模型 agent 的 reward hacking。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://arxiv.org/abs/2605.02964","links":[{"key":"data","label":["Data","数据"],"url":"https://kunvarthaman.com/posts/rhb-v1.html"}],"link_count":2,"sections":9},{"id":"reward-hacking-language-model-agents-2026","title":"Reward Hacking in Language Model Agents: Revisiting AI Safety Gridworlds","year":2026,"venue":"arXiv","authors":["Ömer Veysel Çağatan","Xuandong Zhao"],"authors_zh":"Ömer Veysel Çağatan, Xuandong Zhao","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"污染、验证器失效或奖励投机审计论文","best_for_zh":"需要审计基准污染、评估偏差或奖励投机风险的研究者。","confidence":"high","one_line":["text-based reward-hacking evaluation environments","将 AI Safety Gridworlds 文本化，量化语言模型 agent 在代理奖励下的 specification gaming。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://arxiv.org/abs/2606.15385","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/asparius/verl-agent-safety"}],"link_count":2,"sections":9},{"id":"reward-under-attack-prm-robustness-2026","title":"Reward Under Attack: Analyzing the Robustness and Hackability of Process Reward Models","year":2026,"venue":"ICML 2026","authors":["Rishabh Tiwari","Aditya Tomar","Udbhav Bamba","Monishwaran Maheswaran","Heng Yang","Michael W. Mahoney","Kurt Keutzer","Amir Gholami"],"authors_zh":"Rishabh Tiwari, Aditya Tomar, Udbhav Bamba, Monishwaran Maheswaran, Heng Yang, Michael W. Mahoney, Kurt Keutzer, Amir Gholami","tracks":["preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-reward-modeling","robustness"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程奖励模型鲁棒性评测、攻击审计与公开基准论文","best_for_zh":"需要在将过程奖励模型投入训练、搜索或评测前检查其对逻辑错误、对抗优化和奖励投机的敏感性的研究者。","confidence":"high","one_line":["Reward Under Attack releases PRM-BiasBench and a three-tier audit showing that process rewards can be optimized on invalid mathematical reasoning.","PRM-BiasBench 含数千受控扰动对与三层攻击，反馈奖励变化，揭示通用数学推理奖励的可被利用性。"],"why":"It turns PRM robustness from an assumed property into an executable diagnostic before using a reward as a training signal.","primary_link":"https://arxiv.org/abs/2603.06621","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SqueezeAILab/reward-under-attack"}],"link_count":5,"sections":9},{"id":"dataprm-agentic-data-analysis-2026","title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","year":2026,"venue":"KDD 2026","authors":["Zhisong Qiu","Shuofei Qiao","Kewei Xu","Yuqi Zhu","Lun Du","Ningyu Zhang","Huajun Chen"],"authors_zh":"Zhisong Qiu、Shuofei Qiao、Kewei Xu、Yuqi Zhu、Lun Du、Ningyu Zhang、Huajun Chen","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision"],"verification_contract":["environmental"],"supervision_granularity":["process_reward"],"training_use":["agent_training"],"construction_layer":["reward_verifier_layer"],"domains":["agentic-reasoning"],"tags":["post-training","process-supervision"],"status":"verified","priority":"必读","paper_type_zh":"数据分析智能体环境感知过程奖励模型论文","best_for_zh":"需要把可执行数据分析轨迹转成过程监督、搜索反馈或强化学习奖励的读者。","confidence":"high","one_line":["DataPRM probes execution states and assigns reflection-aware process rewards to train and guide data-analysis agents.","DataPRM 探测执行状态并赋予反思感知的过程奖励，用于训练和引导数据分析智能体。"],"why":"It exposes a process-feedback object and its consumer in post-training.","primary_link":"https://arxiv.org/abs/2604.24198","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zjunlp/DataMind"}],"link_count":2,"sections":9},{"id":"re2-resolving-2026","title":"Re²: Unlocking LLM Reasoning via Reinforcement Learning with Re-solving","year":2026,"venue":"ICLR 2026","authors":["Pinzheng Wang","Shuli Xu","Juntao Li","Yu Luo","Dong Li","Jianye Hao","Min Zhang"],"authors_zh":"Pinzheng Wang、Shuli Xu、Juntao Li、Yu Luo、Dong Li、Jianye Hao、Min Zhang（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning","software-engineering"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["Re²: Unlocking LLM Reasoning via Reinforcement Learning with Re-solving","RLVR 模型可能持续展开一条起点很差的思维链而不放弃它，因此更多 token 造成过度思考而非恢复。"],"why":"It makes a test-time decision auditable.","primary_link":"https://arxiv.org/abs/2603.07197","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PinzhengWang322/rl-resolving"}],"link_count":3,"sections":9},{"id":"rlbff-binary-flexible-feedback-2026","title":"RLBFF: Binary Flexible Feedback to bridge between Human Feedback & Verifiable Rewards","year":2026,"venue":"ICLR 2026","authors":["Zhilin Wang","Jiaqi Zeng","Olivier Delalleau","Ellie Evans","Daniel Egert","Hoo-Chang Shin","Felipe Soares","Yi Dong","Oleksii Kuchaiev"],"authors_zh":"Zhilin Wang, Jiaqi Zeng, Olivier Delalleau, Ellie Evans, Daniel Egert, Hoo-Chang Shin, Felipe Soares, Yi Dong, Oleksii Kuchaiev","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","preference-learning","alignment"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"可解释二元反馈、奖励模型与强化学习对齐论文","best_for_zh":"需要把自然语言反馈转换为可配置的二元原则标签，并训练可按原则进行判断的奖励模型的研究者。","confidence":"high","one_line":["RLBFF turns natural-language feedback into binary principles so reward models can score whether a response satisfies a chosen quality criterion.","4.08万带人类文字反馈的样本被转为二元原则满足标签，可让 RM 按准确性、代码质量等维度推理。"],"why":"It preserves an explicit, configurable feedback contract between open-ended human feedback and reward-model supervision.","primary_link":"https://openreview.net/forum?id=P3R3S6S5Km","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/RL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/HelpSteer3"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/reward-models-10-2025"}],"link_count":6,"sections":9},{"id":"rlvr-datasets-where-find-2026","title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","year":2026,"venue":"arXiv preprint arXiv:2605.26971","authors":["Hsiu-Yuan Huang","Weijie Liu","Chenming Tang","Sanwoo Lee","Kai Yang","Yangkun Chen","Saiyong Yang","Yunfang Wu"],"authors_zh":"Hsiu-Yuan Huang、Weijie Liu、Chenming Tang、Sanwoo Lee、Kai Yang、Yangkun Chen、Saiyong Yang、Yunfang Wu","tracks":["data_construction_open_release_recipes","foundations_and_primers"],"source_role":["data_release","construction_recipe","benchmark","verifier_reward","audit_failure"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["math"],"tags":["rlvr","data-lineage","atomic-source-tracing","source-counterfactual-attribution","dapo-plus","decontamination","math-verify","grpo","dataset-benchmarking","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"RLVR 数据谱系审计与构造配方","best_for_zh":"关注 RLVR 数据来源、污染控制、可学习性归因与开放发布审计的研究者","confidence":"medium","one_line":["ATLAS traces 1.45M RLVR instances to atomic sources and uses leakage and source-utility signals to construct the released 17K-row DAPO++, while the lineage and selection ledgers remain unreleased.","ATLAS 将 145 万条 RLVR 记录追溯到原子来源，并据泄漏与来源效用信号构建公开的 1.7 万条 DAPO++；完整谱系与选择账本尚未发布。"],"why":"It links provenance evidence to an explicit RLVR construction and benchmarking recipe, and shows why a final Parquet cannot support full reuse auditing when atomic-source labels, SCA decisions, random replacement history, licenses, and immutable release metadata are missing.","primary_link":"https://arxiv.org/abs/2605.26971","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Celine-hxy/ATLAS"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Celine-hxy/ATLAS/blob/main/DAPO_Plus.parquet"}],"link_count":4,"sections":9},{"id":"robometer-robotic-reward-trajectories-2026","title":"Robometer: Scaling General-Purpose Robotic Reward Models via Trajectory Comparisons","year":2026,"venue":"Robotics: Science and Systems 2026","authors":["Anthony Liang","Yigit Korkmaz","Jiahui Zhang","Minyoung Hwang","Abrar Anwar","Sidhant Kaushik","Aditya Shah","Alex S. Huang","Luke Zettlemoyer","Dieter Fox","Yu Xiang","Anqi Li","Andreea Bobu","Abhishek Gupta","Stephen Tu","Erdem Bıyık","Jesse Zhang"],"authors_zh":"Anthony Liang, Yigit Korkmaz, Jiahui Zhang, Minyoung Hwang, Abrar Anwar, Sidhant Kaushik, Aditya Shah, Alex S. Huang, Luke Zettlemoyer, Dieter Fox, Yu Xiang, Anqi Li, Andreea Bobu, Abhishek Gupta, Stephen Tu, Erdem Bıyık, Jesse Zhang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["robotics","trajectory-preference","reward-learning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"机器人轨迹比较奖励模型与百万级数据集论文","best_for_zh":"需要从机器人失败、次优与专家轨迹共同学习可泛化奖励函数的研究者。","confidence":"high","one_line":["Robometer learns robot rewards from both per-frame progress and same-task trajectory preferences, including failed executions.","RBM-1M 含逾百万机器人轨迹，结合帧级进度与同任务轨迹偏好，显式纳入失败和次优行为。"],"why":"It makes otherwise ambiguous suboptimal robot data usable for reward learning.","primary_link":"https://arxiv.org/abs/2603.02115","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/robometer/robometer"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/robometer"},{"key":"project","label":["Project","项目主页"],"url":"https://robometer.github.io/"}],"link_count":5,"sections":9},{"id":"roboreward-general-purpose-vision-language-reward-models-for-robotics-2026","title":"RoboReward: General-Purpose Vision-Language Reward Models for Robotics","year":2026,"venue":"arXiv","authors":["Tony Lee","Andrew Wagenmaker","Karl Pertsch","Percy Liang","Sergey Levine","Chelsea Finn"],"authors_zh":"Tony Lee、Andrew Wagenmaker、Karl Pertsch、Percy Liang、Sergey Levine、Chelsea Finn","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","answer_level"],"training_use":["reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["robotics","multimodal","alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"机器人视觉语言奖励模型数据集与基准论文","best_for_zh":"训练或评测机器人视觉语言奖励模型，并研究可复用的轨迹奖励构造。","confidence":"high","one_line":["RoboReward turns real-robot videos into a five-level reward dataset and human-verified benchmark for general-purpose vision-language reward models.","RoboReward 将真实机器人视频转为五级进度奖励数据与人工核验基准，用于训练通用视觉语言奖励模型。"],"why":"It makes a concrete human-feedback or reward object available for alignment research.","primary_link":"https://arxiv.org/abs/2601.00675","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/teetone/RoboReward"},{"key":"project","label":["Project","项目主页"],"url":"https://crfm.stanford.edu/helm/robo-reward-bench/v0.0.1/"}],"link_count":3,"sections":9},{"id":"rubric-arrow-pointwise-reward-modeling-2026","title":"RUBRIC-ARROW: Alternating Pointwise Rubric Reward Modeling for LLM Post-training in Non-verifiable Domains","year":2026,"venue":"arXiv","authors":["Haoxiang Jiang","Zihan Dong","Tianci Liu","Wanying Wang","Ran Xu","Tony Yu","Linjun Zhang","Haoyu Wang"],"authors_zh":"Haoxiang Jiang, Zihan Dong, Tianci Liu, Wanying Wang, Ran Xu, Tony Yu, Linjun Zhang, Haoyu Wang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["general"],"tags":["rubric","reward_model","pointwise","judge","post_training"],"status":"verified","priority":"可读","paper_type_zh":"Rubric 条件化点式奖励模型、Judge SFT 数据与后训练方法","best_for_zh":"需要构建开放式任务的 rubric Judge、点式奖励模型或偏好后训练流程的研究者。","confidence":"high","one_line":["RUBRIC-ARROW alternates rubric generation and pointwise judging to train reward models for non-verifiable tasks.","RUBRIC-ARROW 交替训练 rubric 生成器与条件化点式 Judge，为不可验证任务提供可学习的奖励信号。"],"why":"It exposes a reusable rubric-conditioned judge data object instead of relying only on opaque frontier evaluators.","primary_link":"https://arxiv.org/abs/2605.29156","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenRubrics/RubricARROW-Judge-SFT"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/OpenRubrics/rubricarrow"}],"link_count":4,"sections":9},{"id":"rubricbench-model-rubrics-2026","title":"RubricBench: Aligning Model-Generated Rubrics with Human Standards","year":2026,"venue":"ACL 2026","authors":["Qiyuan Zhang","Junyi Zhou","Yufei Wang","Fuyuan Lyu","Yidong Ming","Can Xu","Qingfeng Sun","Kai Zheng","Peng Kang","Xue Liu","Chen Ma"],"authors_zh":"Qiyuan Zhang, Junyi Zhou, Yufei Wang, Fuyuan Lyu, Yidong Ming 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["rubric_generation","alignment"],"tags":["rubric","reward_model","alignment"],"status":"verified","priority":"必读","paper_type_zh":"模型生成 rubric 与人工标准对齐基准","best_for_zh":"研究 LLM Judge、rubric 奖励模型与对齐评测的读者。","confidence":"medium","one_line":["RubricBench tests whether model-generated evaluation criteria align with expert atomic rubrics.","RubricBench 以专家原子 rubric 和胜负标签检验模型生成准则是否忠实反映人类评判标准。"],"why":"It evaluates the rubric itself, a key source of error in rubric-guided alignment.","primary_link":"https://arxiv.org/abs/2603.01562","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/planepig/rubricbench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/DonJoey/rubricbench"}],"link_count":5,"sections":9},{"id":"rubrichub-coarse-to-fine-rubric-generation-2026","title":"RubricHub: A Comprehensive and Highly Discriminative Rubric Dataset via Automated Coarse-to-Fine Generation","year":2026,"venue":"ACL 2026","authors":["Sunzhu Li","Jiale Zhao","Huimin Ren","Zhenlin Wei","Yang Zhou","Jingwen Yang","Shunyu Liu","Kaike Zhang","Wei Chen"],"authors_zh":"Sunzhu Li, Jiale Zhao, Huimin Ren, Zhenlin Wei, Yang Zhou, Jingwen Yang, Shunyu Liu, Kaike Zhang, Wei Chen","tracks":["preference_reward_feedback_data","judgment_rubric_domain_expert_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["open-ended-generation","rubric-reward-modeling","alignment"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"自动粗到细评分准则生成与跨域奖励数据论文","best_for_zh":"需要为开放式生成构造细粒度、带权重的评分准则，并进行拒绝采样微调或基于准则强化学习的研究者。","confidence":"high","one_line":["RubricHub releases roughly 110K cross-domain, weighted rubric evaluations for fine-tuning and rubric-based RL on open-ended generation.","约 11 万跨域 rubric/评分样本，含 criterion、权重、评分和评审细节，可供 RM 微调与 RL。"],"why":"It preserves the criterion, weight, score, and judge-detail fields needed to audit how an open-ended reward was produced.","primary_link":"https://aclanthology.org/2026.acl-long.1445/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sojuL/RubricHub_v1"}],"link_count":4,"sections":9},{"id":"saferun-deterministic-running-planning-2026","title":"SafeRun: Enabling Determinism in LLM Planning for Running","year":2026,"venue":"ICML 2026 LM4Plan Workshop","authors":["Meilin Chen","Zepeng Zhai","Jiaxuan Zhao","Yuan Lu"],"authors_zh":"Meilin Chen, Zepeng Zhai, Jiaxuan Zhao, Yuan Lu","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["planning","safety"],"tags":["planning","deterministic-solver","safety","benchmark","icml-2026"],"status":"verified","priority":"可读","paper_type_zh":"带确定性安全验证器的跑步规划基准","best_for_zh":"研究硬约束规划、工具调用和安全验证器的研究者。","confidence":"high","one_line":["SafeRun pairs running-plan requests with a CP-SAT solver so every accepted plan is checked against explicit safety constraints.","SafeRun 将跑步计划请求交给 CP-SAT 求解器执行硬约束，使每个被接受的计划都通过明确的安全规则检查。"],"why":"It provides a narrow but fully programmatic planning surface for studying hard-constraint enforcement.","primary_link":"https://arxiv.org/abs/2606.09027","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zzp-seeker/SafeRun-RunPlanning-Benchmark"}],"link_count":3,"sections":9},{"id":"sage-judge-without-human-effort-2026","title":"Sage: A Scalable Framework for Evaluating LLM-as-a-Judge Without Human Effort","year":2026,"venue":"ICLR 2026 submission","authors":["Yuanning Feng","Sinan Wang","Zhengxiang Cheng","Yao Wan","Dongping Chen"],"authors_zh":"Yuanning Feng，Sinan Wang，Zhengxiang Cheng，Yao Wan，Dongping Chen。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"无人工标注的 LLM-as-a-Judge 一致性评测框架","best_for_zh":"需大规模审计自动 judge、但缺少人工金标的团队。","confidence":"medium","one_line":["Introduces consistency-based judge evaluation without human labels; status is explicitly recorded as submission.","用局部偏好稳定性和全局传递性，无需人工标注地审计 LLM judge 的一致性。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=JFTSZa2stt","links":[],"link_count":1,"sections":9},{"id":"saja-2026","title":"SAJA: A Simple Approach to Judge Alignment for LLM-as-a-Judge","year":2026,"venue":"ACL 2026 Industry Track","authors":["Sneha Kola","Pankaj Kumar Sharma","Soumyadeep Dey","Bamdev Mishra","Mayur Datar"],"authors_zh":"Sneha Kola、Pankaj Kumar Sharma、Soumyadeep Dey、Bamdev Mishra、Mayur Datar","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"LLM-as-a-Judge 校准与部署方法论文","best_for_zh":"需要在闭源 API 上以少量人工标签构建可审计、低成本 LLM 评委的研究者和工程团队。","confidence":"medium","one_line":["A rubric-based calibration approach that aligns one-call LLM judgments to a small human-labeled set.","用一次多维 rubric 调用和少量人工标签训练校准头，在不读取 logit 的条件下对齐 LLM 评委。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.acl-industry.45/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/SAJA"}],"link_count":2,"sections":9},{"id":"scaleenv-2026","title":"ScaleEnv: Scaling Environment Synthesis from Scratch for Generalist Interactive Tool-Use Agent Training","year":2026,"venue":"ICML 2026","authors":["Dunwei Tu","Hongyan Hao","Hansi Yang","Yihao Chen","Yi-Kai Zhang","Zhikang Xia","Yu Yang","Yueqing Sun","Xingchen Liu","Furao Shen","Qi Gu","Hui Su","Xunliang Cai"],"authors_zh":"Dunwei Tu, Hongyan Hao, Hansi Yang, Yihao Chen, Yi-Kai Zhang, Zhikang Xia, Yu Yang, Yueqing Sun, Xingchen Liu, Furao Shen, Qi Gu, Hui Su, Xunliang Cai","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","construction_recipe","scaling_study","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","synthetic_environments"],"tags":["scaleenv","environment-synthesis","tool-use","executable-sandbox","database-state","procedural-testing","final-state-verifier","grpo","environment-scaling","replay-risk"],"status":"partial","priority":"必读","paper_type_zh":"可执行环境合成、程序化验证与智能体强化学习","best_for_zh":"研究环境数据工厂、数据库状态奖励、大批量 rollout、OOD 泛化及 reset/replay 审计的读者","confidence":"high","one_line":["ScaleEnv synthesizes 16 database-backed domains and 2,560 verifiable tasks for large-batch GRPO, but the environments, rollouts, rewards, code, checkpoints, reset mechanism, and replay manifests are not verified as released.","ScaleEnv 自动合成 16 个数据库工具域与 2,560 个可验证任务，用于大批量 GRPO；但完整环境、轨迹、奖励、代码、checkpoint、reset 机制与 replay 清单尚未核实公开。"],"why":"It links scalable environment generation to executable verification and OOD agent training while exposing the distinction between a documented recipe and a reusable, licensed, replayable environment/trajectory corpus.","primary_link":"https://openreview.net/forum?id=3975b88aa41ca35985e91cfea741330c1253286d","links":[],"link_count":6,"sections":9},{"id":"selective-tts-unverifiable-rewards-2026","title":"Scaling Unverifiable Rewards: A Case Study on Visual Insights","year":2026,"venue":"Findings of ACL 2026","authors":["Shuyu Gan","James Mooney","Pan Hao","Renxiang Wang","Mingyi Hong","Qianwen Wang","Dongyeop Kang"],"authors_zh":"Shuyu Gan、James Mooney、Pan Hao、Renxiang Wang、Mingyi Hong、Qianwen Wang、Dongyeop Kang（机构：明尼苏达大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-reasoning","data-analysis"],"tags":["test-time-scaling","multi-agent","process-refinement","branch-pruning","unverifiable-reward"],"status":"verified","priority":"可读","paper_type_zh":"多智能体阶段级测试时扩展研究","best_for_zh":"设计长程、多阶段智能体流程，且最终质量带有主观性、必须在最终输出出现前分配计算预算的读者。","confidence":"high","one_line":["Selective TTS uses stage-specific judges to prune weak branches early in a data-analysis agent pipeline, improving visual-insight quality under a fixed compute budget.","选择性测试时扩展使用阶段专属评审器，在数据分析智能体流程中及早剪去弱分支，以固定计算预算提升视觉洞见质量。"],"why":"It shows that test-time scaling need not repeatedly refine one answer: a fixed budget can be distributed across process stages to reduce judge-error propagation.","primary_link":"https://aclanthology.org/2026.findings-acl.1724/","links":[{"key":"code","label":["Code","代码"],"url":"https://minnesotanlp.github.io/insight-scaling-webpage"}],"link_count":3,"sections":9},{"id":"speculative-decoding-tts-benchmark-2026","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","year":2026,"venue":"ICLR 2026","authors":["Shengyin Sun","Yiming Li","Xing Li","Yingzhao Lian","Weizhe Lin","Huiling Zhen","Zhiyuan Yang","Xianzhi Yu","Chen Chen","Mingxuan Yuan","Chen Ma"],"authors_zh":"Shengyin Sun、Yiming Li、Xing Li、Yingzhao Lian、Weizhe Lin、Huiling Zhen、Zhiyuan Yang、Xianzhi Yu、Chen Chen、Mingxuan Yuan、Chen Ma","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","scaling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配的读者。","confidence":"high","one_line":["Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","重复推理轨迹存在冗余，因此额外测试时样本可能让系统过慢。"],"why":"It measures a concrete decision about additional inference computation.","primary_link":"https://arxiv.org/abs/2509.04474","links":[],"link_count":1,"sections":9},{"id":"scaling-confidence-adaptive-tts-2026","title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","year":2026,"venue":"arXiv preprint; ACL ARR 2026 submission","authors":["Xuqing Yang","Yi Yuan","Shanzhe Lei","Xuhong Wang"],"authors_zh":"Xuqing Yang（上海交通大学）；Yi Yuan（东南大学）；Shanzhe Lei、Xuhong Wang（上海人工智能实验室）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","multimodal-reasoning","knowledge-reasoning"],"tags":["test-time-compute","confidence-calibration","adaptive-sampling","early-stopping","rlvr"],"status":"verified","priority":"可读","paper_type_zh":"置信度校准驱动的自适应测试时扩展研究","best_for_zh":"研究如何用可靠不确定性分配推理采样预算的读者。","confidence":"high","one_line":["C3RL trains verbalized confidence to track correctness, then CAS uses that signal to stop easy queries early and reserve sampling for uncertain ones.","C3RL 先训练模型把口头置信度校准到真实正确率，再由 CAS 据此为不确定问题追加采样。"],"why":"It connects a training-time calibration objective to an explicit inference-time allocation rule and reports both quality and saved samples.","primary_link":"https://arxiv.org/abs/2607.01612","links":[],"link_count":3,"sections":9},{"id":"search-arena-analyzing-search-augmented-llms-2026","title":"Search Arena: Analyzing Search-Augmented LLMs","year":2026,"venue":"ICLR 2026","authors":["Mihran Miroyan","Tsung-Han Wu","Logan King","Tianle Li","Jiayi Pan","Xinyan Hu","Wei-Lin Chiang","Anastasios N. Angelopoulos","Trevor Darrell","Narges Norouzi","Joseph E. Gonzalez"],"authors_zh":"Mihran Miroyan 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","full_episode"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["search-augmented-llms","web-grounding","human-preference"],"tags":["search","arena","human-votes","citations","retrieval-traces"],"status":"verified","priority":"可读","paper_type_zh":"搜索增强语言模型的人类偏好数据集与分析论文","best_for_zh":"研究检索增强生成、搜索对话偏好或引用可靠性的读者。","confidence":"high","one_line":["Search Arena releases real multi-turn search-LLM battles with user votes and full retrieval traces, exposing the difference between preferred and well-supported answers.","Search Arena 发布真实搜索增强模型对话、用户偏好票和检索轨迹，可检验偏好与引用支撑是否一致。"],"why":"It joins observed human preferences to the retrieved evidence needed to audit whether citation-rich answers are actually grounded.","primary_link":"https://arxiv.org/abs/2506.05334","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lmarena/search-arena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lmarena-ai/search-arena-24k"},{"key":"project","label":["Project","项目主页"],"url":"https://legacy.lmarena.ai/"}],"link_count":6,"sections":9},{"id":"search-time-contamination-deep-research-2026","title":"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation","year":2026,"venue":"arXiv preprint","authors":["Yongjie Wang","Xinyue Zhang","Kunhong Yao","Zhiwei Zeng","Kaisong Song","Jun Lin","Zhiqi Shen"],"authors_zh":"Yongjie Wang, Xinyue Zhang, Kunhong Yao, Zhiwei Zeng, Kaisong Song, Jun Lin, Zhiqi Shen","tracks":["environment_agent_trajectory_data"],"source_role":["audit_failure","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level"],"training_use":["audit"],"construction_layer":["release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use"],"tags":["environment-agent-trajectory-data","agent-trajectories","deep-research","web-search","contamination-audit","evaluation"],"status":"partial","priority":"必读","paper_type_zh":"Deep-research 搜索轨迹、污染检测与评测审计","best_for_zh":"研究联网智能体证据 provenance、搜索轨迹、benchmark contamination、verifier 与可重放评测的读者","confidence":"high","one_line":["Across 6,803 questions from six medical/clinical benchmarks, the study labels benchmark metadata, question context, and explicit answers inside deep-research search trajectories and shows that direct answer exposure can inflate accuracy and accelerate prediction convergence.","该研究在六个医学/临床基准的 6,803 个问题上逐步标注 deep-research 轨迹中的基准 metadata、问题上下文与显式答案泄漏，表明 web observation 会抬高准确率并加快预测收敛。"],"why":"It demonstrates that web observations are not neutral evidence: they can contain the evaluation item itself, so auditable search trajectories, controlled retrieval corpora, and contamination-aware scoring are necessary for credible agent evaluation.","primary_link":"https://arxiv.org/abs/2606.05241","links":[],"link_count":1,"sections":9},{"id":"secrepobench-secure-code-completion-2026","title":"SecRepoBench: Benchmarking Code Agents for Secure Code Completion in Real-World Repositories","year":2026,"venue":"LLM4Code 2026","authors":["Chihao Shen","Connor Dilgren","Purva Chiniya","Luke Griffith","Yu Ding","Yizheng Chen"],"authors_zh":"Chihao Shen, Connor Dilgren, Purva Chiniya, Luke Griffith, Yu Ding, Yizheng Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-security"],"tags":["secure-code","repository-benchmark","c-cpp","cwe","executable-evaluation"],"status":"verified","priority":"必读","paper_type_zh":"仓库级安全代码补全基准","best_for_zh":"研究安全代码代理、仓库级补全或执行式代码验证的研究者。","confidence":"high","one_line":["SecRepoBench uses paired developer tests and exploit checks on 318 real C/C++ repository holes to measure whether code agents produce both functional and secure completions.","SecRepoBench 在 318 个真实 C/C++ 仓库缺口上以开发者测试与漏洞利用复验共同衡量补全的功能和安全性。"],"why":"It turns security-sensitive completion into a joint executable outcome rather than a text-match or compilation-only task.","primary_link":"https://arxiv.org/abs/2504.21205","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ai-sec-lab/SecRepoBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ai-sec-lab/SecRepoBench"},{"key":"project","label":["Project","项目主页"],"url":"https://secrepobench.github.io/"}],"link_count":6,"sections":9},{"id":"traceelephant-2026","title":"Seeing the Whole Elephant: A Benchmark for Failure Attribution in LLM-based Multi-Agent Systems","year":2026,"venue":"ACL 2026 Long Papers","authors":["Mengzhuo Chen","Junjie Wang","Fangwen Mu","Yawen Wang","Zhe Liu","Huanxiang Feng","Qing Wang"],"authors_zh":"Mengzhuo Chen、Junjie Wang、Fangwen Mu、Yawen Wang、Zhe Liu、Huanxiang Feng、Qing Wang","tracks":["environment_agent_trajectory_data","process_trace_supervision_data"],"source_role":["benchmark","data_release","agent_environment","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","step_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","multi_agent_systems","failure_attribution","web_and_tool_use","software_engineering"],"tags":["traceelephant","agent-failure-attribution","multi-agent-systems","full-observability","failed-trajectories","replay","benchmark-audit"],"status":"partial","priority":"可读","paper_type_zh":"多智能体失败归因基准与轨迹数据集","best_for_zh":"研究智能体轨迹、失败归因、回放评测与数据审计的读者","confidence":"high","one_line":["TraceElephant releases 220 fully observable failed Captain-Agent, Magentic-One, and SWE-Agent episodes with expert component/step attribution and replay code, while archive repair history and environment drift remain material audit limits.","TraceElephant 发布 220 条 Captain-Agent、Magentic-One 与 SWE-Agent 失败 episode，保留可观察的完整输入、输出和工具轨迹并标注负责组件与决定性步骤；但数据包修复历史和环境漂移限制精确回放。"],"why":"It turns agent debugging into a concrete post-training data surface: task outcomes, inputs, tool observations, agent actions, and step-level causal labels are co-located, enabling attribution evaluation without pretending that judgment labels or mutable environments are deterministic training rewards.","primary_link":"https://aclanthology.org/2026.acl-long.912/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TraceElephant/TraceElephant"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TraceElephant/TraceElephant"}],"link_count":6,"sections":9},{"id":"select2reason-long-cot-selection-2026","title":"Select2Reason: Efficient Instruction-Tuning Data Selection for Long-CoT Reasoning","year":2026,"venue":"Findings of ACL 2026","authors":["Cehao Yang","Xueyuan Lin","Xiaojun Wu","Chengjin Xu","Xuhui Jiang","Honghao Liu","Hui Xiong","Jian Guo"],"authors_zh":"Cehao Yang、Xueyuan Lin、Xiaojun Wu、Chengjin Xu、Xuhui Jiang、Honghao Liu、Hui Xiong、Jian Guo","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["sft"],"construction_layer":["trace_writing"],"domains":["mathematical-reasoning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"长链式推理 SFT 数据筛选论文","best_for_zh":"需要在有限 SFT 预算下选择长 CoT 训练样本的读者。","confidence":"high","one_line":["Select2Reason ranks long-CoT examples by model-relative difficulty and normalized trace length to form a compact SFT subset.","Select2Reason 按题目难度和去重后的推理长度排序长 CoT 样本，以更小的 SFT 子集训练推理能力。"],"why":"It makes the link between an explicit data object and a training objective inspectable.","primary_link":"https://aclanthology.org/2026.findings-acl.331/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/open-r1/OpenR1-Math-220k"}],"link_count":4,"sections":9},{"id":"seta-scaling-environments-for-terminal-agents","title":"SETA: Scaling Environments for Terminal Agents","year":2026,"venue":"arXiv","authors":["Qijia Shen","Zhiqi Huang","Vamsidhar Kamanuru","Aznaur Aliev","Jay Rainton","Ahmed Awelkair","Zhichen Zeng","Jiajun Li","Shi Dong","Yueming Yuan","Boyuan Ma","Qizheng Zhang","Jiwei Fu","Yuzhen Mao","Wendong Fan","Ping Nie","Philip Torr","Bernard Ghanem","Changran Hu","Jonathan Lingjie Li","Urmish Thakker","Guohao Li"],"authors_zh":"Qijia Shen, Zhiqi Huang, Vamsidhar Kamanuru, Aznaur Aliev, Jay Rainton, Ahmed Awelkair, Zhichen Zeng, Jiajun Li, Shi Dong, Yueming Yuan, Boyuan Ma, Qizheng Zhang, Jiwei Fu, Yuzhen Mao, Wendong Fan, Ping Nie, Philip Torr, Bernard Ghanem, Changran Hu, Jonathan Lingjie Li, Urmish Thakker, Guohao Li","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A dual Synth/Evol pipeline releasing over 4,500 uniformly verifiable terminal RL environments.","通过 Synth 与 Evol 双流水线发布 4,500+ 个统一可验证终端 RL 环境。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2607.10891","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/camel-ai/seta"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/camel-ai/SETA-Env"},{"key":"project","label":["Project","项目主页"],"url":"https://www.camel-ai.org/blogs/seta-scaling-environments-for-terminal-agents"}],"link_count":4,"sections":9},{"id":"shepherd-2026","title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","year":2026,"venue":"arXiv preprint","authors":["Simon Yu","Derek Chong","Ananjan Nandi","Dilara Soylu","Jiuding Sun","Christopher D Manning","Weiyan Shi"],"authors_zh":"Simon Yu, Derek Chong, Ananjan Nandi, Dilara Soylu, Jiuding Sun, Christopher D Manning, Weiyan Shi","tracks":["environment_agent_trajectory_data"],"source_role":["infrastructure","agent_environment","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward","trajectory_value"],"training_use":["rlvr","agent_training"],"construction_layer":["trace_writing","search_substrate","optimizer_scaffold","reward_verifier_layer"],"domains":["agent_trajectories","tool_use","software_engineering","reinforcement_learning"],"tags":["environment-agent-trajectory-data","agent-trajectories","reversible-execution","meta-agents","tree-grpo","counterfactual-replay"],"status":"partial","priority":"必读","paper_type_zh":"可逆智能体执行底座、元智能体框架与树式强化学习构造方法","best_for_zh":"研究环境交互轨迹、长程智能体强化学习、反事实回放、可复现性与安全审计的读者","confidence":"high","one_line":["Shepherd turns agent execution into a reversible Git-like trace and uses meta-agent-selected forks for Tree-GRPO, while leaving generated trajectories, checkpoints, splits, and data-governance artifacts unreleased.","Shepherd 将智能体动作、工具调用与耦合沙箱状态写成可逆的 Git 式执行轨迹，并以元智能体选点分支构造 Tree-GRPO 监督；但生成轨迹、模型检查点、精确划分和数据治理材料尚未发布。"],"why":"It exposes the state, action, environment, branch, and reward relationships needed for auditable long-horizon agent supervision, while showing that rollback guarantees and dataset release guarantees are separate questions.","primary_link":"https://arxiv.org/abs/2605.10913","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/shepherd-agents/shepherd"},{"key":"project","label":["Project","项目主页"],"url":"https://shepherd-agents.ai/"}],"link_count":7,"sections":9},{"id":"skill-aware-reasoning-distillation-2026","title":"Skill-Aware Data Selection and Fine-Tuning for Data-Efficient Reasoning Distillation","year":2026,"venue":"ACL 2026 Short Papers","authors":["Lechen Zhang","Yunxiang Zhang","Wei Hu","Lu Wang"],"authors_zh":"Lechen Zhang, Yunxiang Zhang, Wei Hu, Lu Wang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold"],"domains":["reasoning","mathematics","instruction-tuning"],"tags":["reasoning-distillation","skill-aware","data-selection","sft"],"status":"verified","priority":"必读","paper_type_zh":"以技能为中心的推理数据选择与蒸馏研究","best_for_zh":"从结构化推理教师语料构建学生自适应子集的读者。","confidence":"high","one_line":["Skill-aware distillation samples teacher traces for the student’s weak skills and adds full skill chains to the SFT target.","技能感知蒸馏按学生薄弱技能抽取教师轨迹，并把完整技能链加入监督微调目标。"],"why":"It makes the relationship between a reasoning record, a student weakness profile, and the SFT consumer explicit.","primary_link":"https://aclanthology.org/2026.acl-short.49/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/orange0629/skill-data-selection"}],"link_count":3,"sections":9},{"id":"skillsbench-2026","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","year":2026,"venue":"arXiv","authors":["Xiangyi Li","Yimin Liu","Wenbo Chen","Bingran You","Zonglin Di","Yifeng He","Shenghan Zheng","Kyoung Whan Choe","Jiankai Sun","Shuyi Wang","Chujun Tao","Binxu Li","Xuandong Zhao","Hejia Geng","Xiaojun Wu","Junwei Zhou","Xiaokun Chen","Hanwen Xing","Yubo Li","Qunhong Zeng","Di Wang","Yuanli Wang","Roey Ben Chaim","Penghao Jiang","Haotian Shen","Luyang Kong","Xinyi Liu","Runhui Wang","Xuanqing Liu","Jiachen Li","Xin Lan","Yueqian Lin","Wengao Ye","Junwei He","Songlin Li","Yue Zhang","Yipeng Gao","Yijiang Li","Ze Ma","Liqiang Jing","Tianyu Wang","Kaixin Li","Yiqi Xue","Haoran Lyu","Yizhuo He","Yuchen Tian","Shutong Wu","Bowei Wang","Yixuan Gao","Bo Chen","Litong Liu","Sikai Cheng","Jiajun Bao","Shuaicheng Tong","Shuwen Xu","Terry Yue Zhuo","Tinghan Ye","Qi Qi","Miao Li","Longtai Liao","Zelin Tan","Chang Shi","Xilin Tang","Srinath Tankasala","Boqin Yuan","Yaoyao Qian","Jianhong Tu","Chenguang Wang","Yizhou Sun","Wei Wang","Aaron Taylor","Ziyue Yang","Changkun Guan","Zhikang Dong","Xinyu Zhang","Steven Dillmann","Han-chung Lee","Dawn Song"],"authors_zh":"Xiangyi Li 等 77 位作者（BenchFlow、OSU、Amazon、UC Berkeley 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["skill-evaluation","model-capabilities"],"tags":["benchmark","capabilities"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / 评测面","best_for_zh":"适合关注智能体技能包、公开任务目录、确定性 verifier、模型 harness 与跨领域能力评测边界的读者。","confidence":"medium_high","one_line":["SkillsBench evaluates model capabilities across skill-labeled tasks.","SkillsBench 以公开任务、技能包和确定性 verifier 评测 Agent Skills 对多领域任务表现的增益。"],"why":"Skill-level benchmarks need transparent task taxonomy and scorer contracts.","primary_link":"https://arxiv.org/abs/2602.12670","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/benchflow-ai/skillsbench"},{"key":"data","label":["Data","数据"],"url":"https://github.com/benchflow-ai/skillsbench/tree/main/tasks"},{"key":"project","label":["Project","项目主页"],"url":"https://www.skillsbench.ai"}],"link_count":5,"sections":9},{"id":"rl-guided-adaptive-sampling-2026","title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","year":2026,"venue":"arXiv preprint","authors":["Runpeng Dai","Tong Zheng","Rui Liu","Chengsong Huang","Hongtu Zhu"],"authors_zh":"Runpeng Dai、Tong Zheng、Rui Liu、Chengsong Huang、Hongtu Zhu（机构：北卡罗来纳大学教堂山分校、马里兰大学帕克分校、圣路易斯华盛顿大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","adaptive-sampling","reinforcement-learning","early-stopping","self-consistency"],"status":"verified","priority":"可读","paper_type_zh":"强化学习引导的自适应测试时采样研究","best_for_zh":"设计成本敏感的停止策略与并行采样策略的推理系统研究者。","confidence":"high","one_line":["RL-Guided Sampling trains a small CPU-deployable policy to stop or expand parallel reasoning samples under explicit cost penalties.","RL-Guided Sampling 用可在 CPU 部署的小型策略网络，决定何时停止或追加并行推理样本。"],"why":"It turns a hand-written sampling budget rule into a learned, auditable decision policy whose quality, latency, and sample costs are measured together.","primary_link":"https://arxiv.org/abs/2606.03102","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RunpengDai/RL-Guided-Adaptive-Sampling"}],"link_count":3,"sections":9},{"id":"smarteval-smart-contracts-2026","title":"SmartEval: A Benchmark for Evaluating LLM-Generated Smart Contracts from Natural Language Specifications","year":2026,"venue":"arXiv","authors":["Abhinav Goel","Agostino Capponi","Alfio Gliozzo","Chaitya Shah"],"authors_zh":"Abhinav Goel, Agostino Capponi, Alfio Gliozzo, Chaitya Shah","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["code","security"],"tags":["smart_contracts","solidity","security","rubric"],"status":"verified","priority":"可读","paper_type_zh":"智能合约生成质量评测与安全审计数据集","best_for_zh":"研究代码生成、智能合约安全或代码质量 judge 的读者。","confidence":"high","one_line":["SmartEval evaluates 9,000 LLM-generated Solidity contracts using references, audits, and a five-dimensional quality rubric.","SmartEval 用专家实现、五维 rubric、审计和编译记录，系统评测 9,000 份 LLM 生成 Solidity 合约的质量。"],"why":"It provides grounded, reproducible quality signals for code-generation and security judges.","primary_link":"https://arxiv.org/abs/2605.09610","links":[{"key":"data","label":["Data","数据"],"url":"https://zenodo.org/records/20046037"}],"link_count":3,"sections":9},{"id":"soft-contamination-means-benchmarks-test-shallow-generalization-2026","title":"Soft Contamination Means Benchmarks Test Shallow Generalization","year":2026,"venue":"arXiv preprint arXiv:2602.12413","authors":[],"authors_zh":"Ari Spiesberger, Juan J. Vazquez, Nicky Pochinkov, Tomáš Gavenčiak, Peli Grietzer, Gavin Leech, Nandi Schoots","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🧭 Surveys and Primers map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","讨论软污染如何让基准看似测到泛化、实则只测到浅层相似性，适合补充传统去重检查。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://arxiv.org/abs/2602.12413","links":[],"link_count":1,"sections":9},{"id":"solve-detect-verify-2025","title":"Solve-Detect-Verify: Inference-Time Scaling with Flexible Generative Verifier","year":2026,"venue":"ACL 2026 Long Papers","authors":["Jianyuan Zhong","Zeju Li","Zhijian Xu","Xiangyu Wen","Kezhi Li","Qiang Xu"],"authors_zh":"Jianyuan Zhong, Zeju Li, Zhijian Xu, Xiangyu Wen, Kezhi Li, Qiang Xu","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["verifier_reward","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","audit"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["reasoning"],"tags":["track5","verifier_selection_audit"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["Solve-Detect-Verify: Inference-Time Scaling with Flexible Generative Verifier records reasoning traces completion probes and verification feedback under fast/slow generative verification.","Solve-Detect-Verify 在快速与慢速生成式验证之间分配预算，并用验证反馈定向修正答案。"],"why":"It makes completion-triggered verification and targeted correction and its audit boundary visible for reasoning-data curation.","primary_link":"https://aclanthology.org/2026.acl-long.2190/","links":[],"link_count":4,"sections":9},{"id":"sorrydb-real-world-lean-theorems-2026","title":"SorryDB: Can AI Provers Complete Real-World Lean Theorems?","year":2026,"venue":"ICML 2026","authors":["Austin Letson","Leopoldo Sarra","Auguste Poiroux","Oliver Dressler","Paul Lezeau","Dhyan Aranha","Frederick Pu","Aaron Hill","Miguel Corredera Hidalgo","Julian Berman","George Tsoukalas","Lenny Taelman"],"authors_zh":"Austin Letson, Leopoldo Sarra, Auguste Poiroux, Oliver Dressler, Paul Lezeau, Dhyan Aranha, Frederick Pu, Aaron Hill, Miguel Corredera Hidalgo, Julian Berman, George Tsoukalas, Lenny Taelman","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","theorem-proving"],"tags":["programmatic-verification","benchmark","2026"],"status":"verified","priority":"可读","paper_type_zh":"动态更新的真实 Lean 证明任务基准","best_for_zh":"需要评测或训练面向真实 Lean 项目的证明智能体的研究者。","confidence":"high","one_line":["SorryDB continuously harvests open Lean proof obligations with pinned repository metadata so proposed proofs can be compiled in their original projects.","SorryDB 从真实 Lean 项目持续抽取待证明命题，并以固定提交、Lean 版本和编译复现信息构成可更新的证明任务库。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://arxiv.org/abs/2603.02668","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SorryDB/SorryDB"},{"key":"project","label":["Project","项目主页"],"url":"https://sorrydb.org/"}],"link_count":4,"sections":9},{"id":"space-self-paced-rft-2026","title":"SPaCe: Unlocking Sample-Efficient Large Language Models Training With Self-Pace Curriculum Learning","year":2026,"venue":"Findings of ACL 2026","authors":["Dai Do","Manh Nguyen","Svetha Venkatesh","Hung Le"],"authors_zh":"Dai Do, Manh Nguyen, Svetha Venkatesh, Hung Le","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics","logic"],"tags":["self-paced-learning","data-reduction","curriculum","rlvr"],"status":"verified","priority":"必读","paper_type_zh":"自定步强化微调的数据选择与调度研究","best_for_zh":"为可验证推理强化微调构建紧凑自适应课程的读者。","confidence":"high","one_line":["SPaCe reduces a reasoning pool by semantic-difficulty clusters and uses a progress-aware bandit to schedule RFT samples.","SPaCe 以语义—难度簇缩减推理池，再用感知进展的老虎机调度强化微调样本。"],"why":"It makes record coverage and changing training exposure jointly explicit in a low-budget RFT policy.","primary_link":"https://aclanthology.org/2026.findings-acl.171/","links":[],"link_count":2,"sections":9},{"id":"spatialreward-bridging-the-perception-gap-in-online-rl-for-image-editing-via-explicit-spatial-reason-2026","title":"SpatialReward: Bridging the Perception Gap in Online RL for Image Editing via Explicit Spatial Reasoning","year":2026,"venue":"ICML 2026","authors":["Yancheng Long","Yankai Yang","Hongyang Wei","Wei Chen","Tianke Zhang","Haonan Fan","Changyi Liu","Kaiyu Jiang","Jiankang Chen","Kaiyu Tang","Bin Wen","Fan Yang","Tingting Gao","Han Li","Shuo Yang"],"authors_zh":"Yancheng Long、Yankai Yang、Hongyang Wei、Wei Chen、Tianke Zhang 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","SpatialReward 将图像编辑要求锚定到预测的空间区域，用显式空间推理提供细粒度在线强化学习奖励。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2602.07458","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lorangan-ddup/SpatialReward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SpatialReward/SpatialReward-Train"},{"key":"project","label":["Project","项目主页"],"url":"https://lorangan-ddup.github.io/SpatialReward/"}],"link_count":4,"sections":9},{"id":"spiral-2025","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","year":2026,"venue":"ICLR 2026","authors":["Bo Liu","Simon Yu","Zichen Liu","Leon Guertler","Penghui Qi","Daniel Balcells","Mickel Liu","Cheston Tan","Weiyan Shi","Min Lin","Wee Sun Lee","Natasha Jaques"],"authors_zh":"Bo Liu、Simon Yu、Zichen Liu、Leon Guertler、Penghui Qi、Daniel Balcells、Mickel Liu、Cheston Tan、Weiyan Shi、Min Lin、Wee Sun Lee、Natasha Jaques","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","agent_environment","data_release"],"verification_contract":["environmental"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["agent_training","rlvr"],"construction_layer":["search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["games","strategic_reasoning","mathematics","general_reasoning"],"tags":["spiral","self-play","multi-agent","multi-turn","zero-sum-games","rae","iclr-2026"],"status":"partial","priority":"可读","paper_type_zh":"多智能体自博弈环境与在线轨迹构建方法","best_for_zh":"关注环境式 reasoning data、在线课程、多智能体 RL、RLVR 与 rollout 审计的研究者","confidence":"high","one_line":["Generates an evolving online stream of multi-turn reasoning and action trajectories through shared-policy self-play in rule-scored zero-sum games.","在规则可评分的零和游戏中让同一策略扮演双方，在线生成多轮推理与动作轨迹；其价值是把环境反馈变成数据构造循环，但并未冻结可审计的在线 rollout 数据。"],"why":"It turns game rules, transitions, and terminal outcomes into a post-training data construction loop while exposing the difference between an open recipe and a frozen, auditable rollout release.","primary_link":"https://arxiv.org/abs/2506.24119","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/spiral-rl/spiral"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/spiral-rl/Spiral-Kuhn-Poker-Qwen3-32B-SFT"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/spiral-rl/spiral"}],"link_count":6,"sections":9},{"id":"spreadsheetbench-2-business-workflows-2026","title":"SpreadsheetBench 2: Evaluating Agents on End-to-End Business Spreadsheet Workflows","year":2026,"venue":"arXiv preprint","authors":["Jian Zhu","Yuzheng Zhang","Zeyao Ma","Bohan Zhang","Armin Schoepf","Daniel Woloch","Peter Yiliu Wang","Guangyu Robert Yang","Samuel Jacob","Siddharth Nagisetty","Abhiram Chundru","Jean Lin","Spencer Mateega","Jing Zhang"],"authors_zh":"Jian Zhu 等（Renmin University of China）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["spreadsheet-agents","business-workflows","office-agents"],"tags":["agent_environment","trajectory_data","spreadsheet-agents","business-workflows","office-agents"],"status":"verified","priority":"可读","paper_type_zh":"arXiv 的商业电子表格工作流评测基准","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"medium","one_line":["SpreadsheetBench 2 evaluates agents on end-to-end business spreadsheet workflows.","SpreadsheetBench 2 用 321 个端到端业务电子表格工作流评测智能体。"],"why":"it extends spreadsheet agents to end-to-end business workflows","primary_link":"https://arxiv.org/abs/2606.29955","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCKBReasoning/SpreadsheetBench-2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/KAKA22/SpreadsheetBench-v2"},{"key":"project","label":["Project","项目主页"],"url":"https://spreadsheetbench.github.io/"}],"link_count":5,"sections":9},{"id":"spurious-rewards-paradox-mechanistically-understanding-how-rlvr-activates-memori-2026","title":"Spurious Rewards Paradox: Mechanistically Understanding How RLVR Activates Memorization Shortcuts in LLMs","year":2026,"venue":"ICML 2026","authors":["Lecheng Yan","Ruizhe Li","Guanhua Chen","Qing Li","Jiahui Geng","Wenxi Li","Longyue Wang","Chenyang Lyu"],"authors_zh":"Lecheng Yan, Ruizhe Li, Guanhua Chen, Qing Li, Jiahui Geng, Wenxi Li, Longyue Wang, Chenyang Lyu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","optimizer_scaffold"],"domains":["math","rlvr","mechanistic-interpretability"],"tags":["RLVR","contamination","mechanistic-interpretability","path-patching","ICML-2026"],"status":"verified","priority":"必读","paper_type_zh":"ICML 2026 的 RLVR 污染与机制可解释性审计论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Mechanistic audit showing that spurious RLVR activates a Qwen memorization shortcut through middle-layer anchors and later structural adapters.","机制审计显示，伪 RLVR 可经中层 Anchor 与后层 Adapter 激活 Qwen 的污染记忆捷径，而非稳定提升推理。"],"why":"It turns suspicious RLVR gains into testable internal hypotheses and supplies a selective offline shortcut-dependence audit.","primary_link":"https://arxiv.org/abs/2601.11061","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/idwts/How-RLVR-Activates-Memorization-Shortcuts"}],"link_count":2,"sections":9},{"id":"srdetection-2026","title":"SrDetection: Self-Referential Leakage Detection for Code Large Language Models","year":2026,"venue":"Findings of ACL 2026","authors":[],"authors_zh":"Shuaimin Li, Liyang Fan, Zeyang Li, Zhuoyue Wan, Yufang Lin, Shiwen Ni, Feiteng Fang, Hamid Alinejad-Rokny, Yuanfeng Song, Kun Jing, Chen Jason Zhang, Min Yang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"代码基准污染与成员推断审计论文","best_for_zh":"维护代码榜单或需要审计闭源 Code LLM 泄漏的研究者。","confidence":"high","one_line":["controlled leakage testbed and self-referential detection code","以语义等价代码变体为内部参照，无阈值检测 Code LLM 的逐题预训练泄漏。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://aclanthology.org/volumes/2026.findings-acl/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SMinL/SrDetectionCode"}],"link_count":2,"sections":9},{"id":"strategic-scaling-test-time-compute-2025","title":"Strategic Scaling of Test-Time Compute: A Bandit Learning Approach","year":2026,"venue":"ICLR 2026","authors":["Bowen Zuo","Yinglun Zhu"],"authors_zh":"Bowen Zuo；Yinglun Zhu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","process_reward","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","bandit","global-allocation","strategic-scaling","mixed-feedback"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A global bandit allocation framework for allocating test-time reasoning compute across instances.","该研究把不同问题间的测试时算力分配建模为 bandit 学习，在线估计难度并优先把预算给可解的困难问题。"],"why":"It makes budget allocation a first-class trace field and warns that a single claimed feedback contract can conceal task-dependent judges and verifiers.","primary_link":"https://arxiv.org/abs/2506.12721","links":[],"link_count":1,"sections":9},{"id":"swe-factory-2026","title":"SWE Data Construction, Automatically!","year":2026,"venue":"ACM FSE 2026 Research Papers","authors":["Lianghong Guo","Yanlin Wang","Caihua Li","Wei Tao","Pengyu Yang","Jiachi Chen","Haoyu Song","Duyu Tang","Zibin Zheng"],"authors_zh":"Lianghong Guo、Yanlin Wang、Caihua Li、Wei Tao、Pengyu Yang、Jiachi Chen、Haoyu Song、Duyu Tang、Zibin Zheng","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","benchmark","verifier_reward","agent_environment","model_report"],"verification_contract":["environmental","programmatic","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["code_reasoning","software_engineering","issue_resolution","coding_agents","repository_agents"],"tags":["swe-factory","github-issues","issue-resolution","coding-agents","software-engineering","docker-environments","multi-agent-construction","fail-to-pass","exit-code-verifier","gold-patch","agent-trajectories","full-parameter-sft","swe-bench","binary-test-resources","contamination-audit","container-reproducibility","license-risk","release-completeness"],"status":"partial","priority":"必读","paper_type_zh":"软件工程任务与 agent 轨迹自动构建、环境验证及开放发布","best_for_zh":"研究 GitHub issue 数据构建、repository-agent SFT、程序化环境 verifier，以及容器、谱系、污染和权利审计","confidence":"high","one_line":["SWE-Factory converts issue and pull-request pairs into runnable coding-agent tasks through binary-test recovery, four-agent Docker and test-script generation, execution feedback, and gold-patch fail-to-pass validation, then samples Kimi-K2 trajectories for supervised fine-tuning.","SWE-Factory 以四代理生成 Dockerfile 与评测脚本，并用 gold-patch fail-to-pass 退出码契约验证 issue-resolution 环境；它公开 671 条任务、2,809 条 messages-only 轨迹和 430 条 Gym 记录，但三者缺少逐行谱系与不可变环境绑定。"],"why":"It demonstrates a bridge from real software changes to executable agent supervision while showing why verifier semantics, immutable containers, source rights, contamination checks, and row-level release lineage are part of data quality rather than optional documentation.","primary_link":"https://conf.researchr.org/details/fse-2026/fse-2026-research-papers/70/SWE-Data-Construction-Automatically-","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DeepSoftwareAnalytics/swe-factory"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-Factory/DeepSWE-Agent-Kimi-K2-Trajectories-2.8K"}],"link_count":10,"sections":9},{"id":"swe-ci-2026","title":"SWE-CI: Evaluating Agent Capabilities in Maintaining Codebases via Continuous Integration","year":2026,"venue":"arXiv preprint","authors":["Jialong Chen","Xander Xu","Hu Wei","Chuan Chen","Bing Zhao"],"authors_zh":"Jialong Chen, Xander Xu, Hu Wei, Chuan Chen, Bing Zhao","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","software_engineering","code_agents","long_horizon_tasks","continuous_integration","benchmark_evaluation"],"tags":["environment-agent-trajectory-data","software-engineering-agents","continuous-integration","long-horizon-maintenance","architect-programmer","pytest-verifier","executable-feedback","trajectory-release","docker-environment","replay-risk","contamination","security"],"status":"partial","priority":"必读","paper_type_zh":"持续集成驱动的长时程代码维护基准与轨迹发布","best_for_zh":"关注软件工程智能体、CI 反馈、可执行奖励、失败轨迹、容器回放与公开轨迹复用边界的读者","confidence":"high","one_line":["SWE-CI evaluates 100 long-horizon repository evolutions from 68 public Python repositories through a 20-epoch Architect-Programmer CI loop and now releases tasks, Docker environments, code, and 115 GB of trajectories, with material split-version, verifier, completeness, replay, rights, privacy, and secret risks.","SWE-CI 以 68 个公开 Python 仓库中的 100 个长期演化任务构造最长 20 轮 Architect–Programmer CI 轨迹，并已发布任务、Docker 环境、代码及 115 GB 轨迹；但版本漂移、测试验证器、归档完整性、回放、权利、隐私和凭据隔离仍需审计。"],"why":"It turns future-test feedback into iterative requirements, repository edits, executable outcomes, and a trajectory-level maintainability score rather than a one-shot patch verdict. The official trajectory release makes process analysis possible, but users must distinguish paper v1 from the moving v2 release and audit test quality, run health, archive coverage, licensing, sanitization, replay, and credential isolation before reuse.","primary_link":"https://arxiv.org/abs/2603.03823","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SKYLENAGE-AI/SWE-CI"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/skylenage-ai/SWE-CI"}],"link_count":11,"sections":9},{"id":"swe-factory-your-automated-factory-for-issue-resolution-training-data-and-evaluation-ben","title":"SWE-Factory: Your Automated Factory for Issue Resolution Training Data and Evaluation Benchmarks","year":2026,"venue":"FSE 2026","authors":["Lianghong Guo","Yanlin Wang","Caihua Li","Wei Tao","Pengyu Yang","Jiachi Chen","Haoyu Song","Duyu Tang","Zibin Zheng"],"authors_zh":"Lianghong Guo, Yanlin Wang, Caihua Li, Wei Tao, Pengyu Yang, Jiachi Chen, Haoyu Song, Duyu Tang, Zibin Zheng","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** SWE-Factory automates environments, exit-code grading, and fail-to-pass validation and releases a 2,809-task Gym.","SWE-Factory 自动完成环境、退出码判分和 F2P 验证，并发布 2,809 条 Gym 任务。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2506.10954","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DeepSoftwareAnalytics/swe-factory"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-Factory/DeepSWE-Agent-Kimi-K2-Trajectories-2.8K"}],"link_count":4,"sections":9},{"id":"swe-fficiency-real-workloads-2025","title":"SWE-fficiency: Can Language Models Optimize Real-World Repositories on Real Workloads?","year":2026,"venue":"ICML 2026","authors":["Jeffrey Jian Ma","Milad Hashemi","Amir Yazdanbakhsh","Kevin Swersky","Ofir Press","Enhui Li","Vijay Janapa Reddi","Parthasarathy Ranganathan"],"authors_zh":"Jeffrey Jian Ma, Milad Hashemi, Amir Yazdanbakhsh, Kevin Swersky, Ofir Press, Enhui Li, Vijay Janapa Reddi, Parthasarathy Ranganathan","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","agent_training","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","performance-engineering"],"tags":["performance-optimization","software-engineering","workloads","2026"],"status":"verified","priority":"可读","paper_type_zh":"可程序验证的仓库级性能优化基准","best_for_zh":"需要以正确性测试和可复现运行时共同评测代码智能体的研究者。","confidence":"high","one_line":["SWE-fficiency tests performance engineering on 498 real repository workloads, requiring an agent to preserve correctness while matching expert speedups.","SWE-fficiency 用 498 个真实工作负载任务衡量智能体能否在通过单测的同时达到或逼近专家性能加速。"],"why":"It evaluates a patch through executable behavior and measured runtime rather than a text-only judgment.","primary_link":"https://arxiv.org/abs/2511.06090","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/swefficiency/swefficiency"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/swefficiency/swefficiency"},{"key":"project","label":["Project","项目主页"],"url":"https://swefficiency.com/"}],"link_count":5,"sections":9},{"id":"swe-perf-repository-performance-optimization-2025","title":"SWE-Perf: Can Language Models Optimize Code Performance on Real-World Repositories?","year":2026,"venue":"ICML 2026","authors":["Xinyi He","Qian Liu","Mingzhe Du","Lin Yan","Zhijie Fan","Yiming Huang","Zejian Yuan","Zejun Ma"],"authors_zh":"Xinyi He, Qian Liu, Mingzhe Du, Lin Yan, Zhijie Fan, Yiming Huang, Zejian Yuan, Zejun Ma","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","agent_training","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","performance-engineering"],"tags":["performance-optimization","software-engineering","runtime","2026"],"status":"verified","priority":"可读","paper_type_zh":"可程序验证的仓库级性能优化基准","best_for_zh":"需要以正确性测试和可复现运行时共同评测代码智能体的研究者。","confidence":"high","one_line":["SWE-Perf evaluates repository-level performance patches on 140 real pull-request tasks using executable correctness tests and runtime measurements.","SWE-Perf 从真实性能优化 PR 构建 140 个仓库级任务，并以正确性测试和运行时间共同验证补丁。"],"why":"It evaluates a patch through executable behavior and measured runtime rather than a text-only judgment.","primary_link":"https://arxiv.org/abs/2507.12415","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/swe-perf/swe-perf"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-Perf/SWE-Perf"},{"key":"project","label":["Project","项目主页"],"url":"https://swe-perf.github.io/"}],"link_count":5,"sections":9},{"id":"swe-rebench-v2-2026","title":"SWE-rebench V2: Language-Agnostic SWE Task Collection at Scale","year":2026,"venue":"ICML 2026","authors":["Ibragim Badertdinov","Maksim Nekrashevich","Anton Shevtsov","Alexander Golubev"],"authors_zh":"Ibragim Badertdinov, Maksim Nekrashevich, Anton Shevtsov, Alexander Golubev","tracks":["environment_agent_trajectory_data","data_construction_open_release_recipes","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","agent_environment"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["full_episode"],"training_use":["agent_training","rlvr","sft","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","release_audit"],"domains":["software_engineering","code","repository_agents","multilingual_programming"],"tags":["swe","repository-agents","executable-environments","test-verification","multilingual-code","rl-environments","data-construction","open-release","artifact-verified"],"status":"verified","priority":"必读","paper_type_zh":"软件工程任务数据发布、构造配方与可执行智能体环境","best_for_zh":"研究仓库智能体训练、RLVR、软件工程环境构造与测试型 verifier 审计的读者","confidence":"high","one_line":["SWE-rebench V2 releases 32,079 prebuilt, test-verifiable repository repair environments across 20 languages plus 126,300 PR-derived training tasks, built by an automated setup, oracle-extraction, LLM-filtering, and metadata pipeline.","SWE-rebench V2 发布 32,079 个跨 20 种语言的预构建可执行仓库修复环境及 126,300 个 PR 衍生训练任务，并公开测试终态反馈与自动构造配方，但不可变回放、污染、权利与失败记录仍不完整。"],"why":"It exposes the environment and terminal test contract needed to turn real repository changes into post-training tasks, while also surfacing the parser, specification, licensing, and replay risks that determine whether those tasks are trustworthy learning signals.","primary_link":"https://arxiv.org/abs/2602.23866","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-rebench/SWE-rebench-V2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nebius/SWE-rebench-V2"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nebius/swe-rebench-v2"}],"link_count":11,"sections":9},{"id":"swe-shepherd-advancing-prms-reinforcing-code-agents-2026","title":"SWE-Shepherd: Advancing PRMs for Reinforcing Code Agents","year":2026,"venue":"arXiv","authors":["Mahir Labib Dihan","Md Ashrafur Rahman Khan"],"authors_zh":"Mahir Labib Dihan、Md Ashrafur Rahman Khan","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["code_agents","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"代码智能体过程奖励模型与轨迹监督数据论文","best_for_zh":"训练或评测代码智能体的中间步骤奖励、搜索和轨迹筛选。","confidence":"high","one_line":["SWE-Shepherd pairs code-agent trajectories with process-level rewards to reinforce intermediate engineering decisions rather than only final issue resolution.","SWE-Shepherd 将代码智能体轨迹与过程级奖励配对，监督中间工程决策而非只看最终问题是否解决。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2604.10493","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mahirlabibdihan/swe-shepherd"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mahirlabibdihan/swe-prm-collection"}],"link_count":3,"sections":9},{"id":"swe-world-2026","title":"SWE-World: Building Software Engineering Agents in Docker-Free Environments","year":2026,"venue":"arXiv preprint","authors":["Shuang Sun","Huatong Song","Lisheng Huang","Jinhao Jiang","Ran Le","Zhihao Lv","Zongchao Chen","Yiwen Hu","Wenyang Luo","Wayne Xin Zhao","Yang Song","Hongteng Xu","Tao Zhang","Ji-Rong Wen"],"authors_zh":"Shuang Sun、Huatong Song、Lisheng Huang、Jinhao Jiang、Ran Le、Zhihao Lv、Zongchao Chen、Yiwen Hu、Wenyang Luo、Wayne Xin Zhao、Yang Song、Hongteng Xu、Tao Zhang、Ji-Rong Wen","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","reward_modeling","agent_training","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["software_engineering","repository_level_code","agentic_reasoning"],"tags":["swe-world","software-engineering-agents","learned-environment","world-model","transition-model","reward-model","docker-free","agent-trajectories","reward-hacking","swe-bench-verified"],"status":"partial","priority":"必读","paper_type_zh":"软件工程智能体环境、轨迹构造与学习式 verifier 方法","best_for_zh":"研究 repository-level agent 轨迹、学习式环境反馈、agent SFT/RL 和 reward hacking 审计的读者","confidence":"high","one_line":["SWE-World trains a step-feedback simulator and a virtual test runner from Docker rollouts, then uses them to generate and reward repository-agent episodes without Docker, while the complete task and trajectory corpora remain unreleased.","SWE-World 用 Docker 轨迹训练 step-level SWT 与 episode-level SWR，在确定性文件系统沙箱中模拟执行反馈和终局测试奖励；完整任务与轨迹语料尚未发布。"],"why":"It exposes a concrete alternative feedback contract for SWE-agent post-training and shows both its scaling value and its audit hazard: learned execution can support SFT, RL and TTS, but imperfect rewards, gold-patch conditioning and missing release lineage can change what behavior is actually reinforced.","primary_link":"https://arxiv.org/abs/2602.03419","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCAIBox/SWE-World"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/RUC-AIBOX/swe-agent-series"}],"link_count":4,"sections":9},{"id":"t1-tool-verification-2026","title":"T1: Tool-integrated Verification for Test-time Compute Scaling in Small Language Models","year":2026,"venue":"ICLR 2026","authors":["Minki Kang","Jongwon Jeong","Jaewoong Cho"],"authors_zh":"Minki Kang（KAIST）、Jongwon Jeong（威斯康星大学麦迪逊分校）、Jaewoong Cho（KRAFTON）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["verifier_reward","scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["reward_modeling","test_time_compute"],"construction_layer":["optimizer_scaffold"],"domains":["mathematical-reasoning","knowledge-intensive-reasoning"],"tags":["test-time-compute","verification","process-reward-model","tools","small-language-models"],"status":"verified","priority":"必读","paper_type_zh":"工具整合验证器与测试时计算扩展论文（ICLR 2026）","best_for_zh":"设计高效验证器，或在工具检查与语言模型打分间分配推理预算的研究者。","confidence":"high","one_line":["T1 uses tools to remove memorization-heavy verification work before a small verifier ranks candidates under test-time scaling.","T1 先用工具完成记忆负担重的核验，再由小型验证器进行最终判别，从而提高有限测试时计算下的候选选择效率。"],"why":"It treats verification capacity as part of the inference budget, rather than assuming a stronger verifier is free.","primary_link":"https://arxiv.org/abs/2504.04718","links":[],"link_count":3,"sections":9},{"id":"terminal-bench-command-line-agents-2026","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","year":2026,"venue":"ICLR 2026 / arXiv","authors":["Mike A. Merrill","Alexander G. Shaw","Nicholas Carlini","Boxuan Li","Harsh Raj","Ivan Bercovich","Lin Shi","Jeong Yeon Shin","Thomas Walshe","E. Kelly Buchanan","Alex Dimakis","Andy Konwinski","Ludwig Schmidt"],"authors_zh":"Mike A. Merrill 等（Stanford University）","tracks":["benchmarks_evaluation_surfaces","programmatically_verifiable_outcome_data"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["terminal-agents","command-line-environments","execution-feedback"],"tags":["agent_environment","trajectory_data","terminal-agents","command-line-environments","execution-feedback"],"status":"verified","priority":"必读","paper_type_zh":"ICLR 2026 / arXiv 的 command-line agent benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["Terminal-Bench evaluates agents on hard realistic command-line tasks with execution feedback.","Terminal-Bench 用真实困难命令行任务和执行反馈评测智能体。"],"why":"terminal predicates and command workflows are core environment data","primary_link":"https://arxiv.org/abs/2601.11868","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/laude-institute/terminal-bench"},{"key":"data","label":["Data","数据"],"url":"https://www.tbench.ai/"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/harborframework/terminal-bench-2.0"}],"link_count":6,"sections":9},{"id":"terminalworld-2026","title":"TerminalWorld: Benchmarking Agents on Real-World Terminal Tasks","year":2026,"venue":"arXiv preprint","authors":["Zhaoyang Chu","Jiarui Hu","Xingyu Jiang","Pengyu Zou","Han Li","Chao Peng","Peter O'Hearn","Earl T. Barr","Mark Harman","Federica Sarro","He Ye"],"authors_zh":"Zhaoyang Chu, Jiarui Hu, Xingyu Jiang, Pengyu Zou, Han Li, Chao Peng, Peter O'Hearn, Earl T. Barr, Mark Harman, Federica Sarro, He Ye","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release","verifier_reward","agent_environment","construction_recipe"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["terminal_agents","shell_and_cli","docker_environments","agent_trajectories","state_based_verification"],"tags":["terminalworld","terminal-agents","docker","harbor","executable-benchmark","state-based-tests","agent-trajectories","programmatic-verification","public-terminal-recordings","evaluation-only"],"status":"partial","priority":"可读","paper_type_zh":"终端智能体可执行基准、环境与 verifier 发布","best_for_zh":"研究终端智能体评测、环境反馈契约、状态测试与可复现性审计的读者","confidence":"medium","one_line":["TerminalWorld converts public terminal recordings into 1,530 Docker-backed CLI tasks with reference solutions and state-based tests, including a manually reviewed 200-task evaluation subset.","TerminalWorld 将公开终端录制转换为 1,530 个带 Docker 环境、参考解答与状态测试的 CLI 任务，并提供人工复核的 200-task 评测子集；该发布仅适合评测，且版本、谱系与许可证边界仍未解决。"],"why":"It provides a concrete environmental feedback contract for terminal agents and shows that successful behavior can diverge sharply from a human command trace, while its mutable manifests, legacy images, removed source links, and conflicting licenses demonstrate why executable benchmarks still need rigorous release audits.","primary_link":"https://arxiv.org/abs/2605.22535","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EuniAI/TerminalWorld"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/EuniAI/TerminalWorld"},{"key":"project","label":["Project","项目主页"],"url":"https://terminalworld.ai/"}],"link_count":6,"sections":9},{"id":"tta-star-small-model-reasoning-2026","title":"Test-Time Scaling for Multistep Reasoning in Small Language Models via A* Search","year":2026,"venue":"ICLR 2026 submission","authors":["Alexander Braverman","Weitong Zhang","Quanquan Gu"],"authors_zh":"Alexander Braverman（康奈尔大学）；Weitong Zhang（北卡罗来纳大学教堂山分校）；Quanquan Gu（加州大学洛杉矶分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","a-star-search","small-language-models","self-reflection","mathematical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"面向小语言模型的无训练 A* 测试时扩展研究","best_for_zh":"设计资源受限推理搜索、且不希望依赖外部过程奖励模型的读者。","confidence":"medium","one_line":["TTA* uses a training-free A*-style search over partial solutions so a small model can spend test-time compute on the most promising self-refined branches.","TTA* 以 A* 风格优先级搜索部分解答，让小语言模型在测试时把计算投入最有希望的自我修订分支。"],"why":"It makes the search-tree priority and self-reflection heuristic explicit, exposing a different allocation mechanism from repeated full-solution sampling.","primary_link":"https://openreview.net/forum?id=eJ1yDj6vtH","links":[],"link_count":1,"sections":9},{"id":"train-to-test-scaling-2026","title":"Test-Time Scaling Makes Overtraining Compute-Optimal","year":2026,"venue":"arXiv preprint","authors":["Nicholas Roberts","Sungjun Cho","Zhiqi Gao","Tzu-Heng Huang","Albert Wu","Gabriel Orlanski","Avi Trost","Kelly Buchanan","Aws Albarghouthi","Frederic Sala"],"authors_zh":"Nicholas Roberts 等（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["scaling-laws","pass-at-k","test-time-compute","overtraining"],"status":"verified","priority":"必读","paper_type_zh":"训练—推理联合标度律论文","best_for_zh":"需要根据重复推理需求选择预训练规模的读者。","confidence":"high","one_line":["Train-to-Test scaling laws jointly optimize parameter count, training tokens, and inference samples under an end-to-end budget.","Train-to-Test 标度律在端到端预算下联合优化参数规模、训练 token 和推理采样数。"],"why":"It makes test-time sampling cost alter the compute-optimal training regime.","primary_link":"https://arxiv.org/abs/2604.01411","links":[],"link_count":2,"sections":9},{"id":"tts-reasoning-machine-translation-2026","title":"Test-Time Scaling of Reasoning Models for Machine Translation","year":2026,"venue":"EACL 2026","authors":["Zihao Li","Shaoxiong Ji","Jörg Tiedemann"],"authors_zh":"Zihao Li、Shaoxiong Ji、Jörg Tiedemann（机构：University of Helsinki、ELLIS Institute Finland、University of Turku）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["machine-translation"],"tags":["test-time-compute","machine-translation","budget-allocation","self-correction","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"跨领域测试时扩展评测研究（EACL 2026）","best_for_zh":"检验推理预算能否迁移到数学和代码之外任务的读者。","confidence":"high","one_line":["This study shows that test-time reasoning helps translation mainly with task-specialized models and post-editing, not by indiscriminately lengthening thought.","该研究表明，测试时推理在翻译中主要通过任务专化模型与后编辑发挥作用，而非无差别延长思考。"],"why":"It identifies boundary conditions under which more inference computation becomes useful rather than wasteful.","primary_link":"https://aclanthology.org/2026.eacl-long.133/","links":[],"link_count":2,"sections":9},{"id":"reflective-generative-model-2026","title":"Test-Time Scaling with Reflective Generative Model","year":2026,"venue":"ICLR 2026","authors":["Zixiao Wang","Yuxin Wang","Xiaorui Wang","Mengting Xing","Jie Gao","Jianjun Xu","Guangcan Liu","Chenhui Jin","Zhuo Wang","Shengzhuo Zhang","Hongtao Xie"],"authors_zh":"Zixiao Wang 等（中国科学技术大学、MetaStone Technology）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","process_reward"],"training_use":["process_supervision","test_time_compute"],"construction_layer":["reward_verifier_layer","optimizer_scaffold"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","process-reward-model","trajectory-selection","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"反思生成与过程验证器效率研究（ICLR 2026）","best_for_zh":"研究轨迹选择、过程奖励模型和推理预算分配的读者。","confidence":"high","one_line":["RGM shares a policy backbone with a self-supervised process scorer to select trajectories under controllable test-time reasoning length.","RGM 让策略模型与自监督过程评分器共享主干，在可控思考长度下选择高质量推理轨迹。"],"why":"It evaluates a small process-scoring head as an alternative to a separate large reward model for test-time scaling.","primary_link":"https://arxiv.org/abs/2507.01951","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MetaStone-AI/MetaStone-S1"}],"link_count":4,"sections":9},{"id":"optimal-transport-test-time-verification-2026","title":"Test-time Verification via Optimal Transport: Coverage, ROC, & Sub-optimality","year":2026,"venue":"ICLR 2026","authors":["Arpan Mukherjee","Marcello Bullo","Debabrota Basu","Deniz Gündüz"],"authors_zh":"Arpan Mukherjee、Marcello Bullo、Debabrota Basu、Deniz Gündüz（机构：帝国理工学院、Université de Lille、Inria、CNRS）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","verifier","optimal-transport","sampling","theory"],"status":"verified","priority":"必读","paper_type_zh":"测试时验证的理论与实证研究（ICLR 2026）","best_for_zh":"设计或审计验证器引导的候选生成与选择流程的读者。","confidence":"high","one_line":["This work models verifier-guided test-time scaling as optimal transport, exposing coverage, verifier, and sampling trade-offs.","该工作将验证器引导的推理时扩展建模为最优传输，揭示覆盖率、验证器与采样之间的取舍。"],"why":"It explains why extra samples can help, stop helping, or hurt when verification is imperfect.","primary_link":"https://openreview.net/pdf/3b8ca7276952cb6b3b2a12f0ae5794d8de2e819d.pdf","links":[],"link_count":4,"sections":9},{"id":"impact-posttraining-contamination-2026","title":"The Impact of Post-training on Data Contamination","year":2026,"venue":"NeurIPS 2025","authors":["Muhammed Yusuf Kocyigit","Caglar Yildirim"],"authors_zh":"Muhammed Yusuf Kocyigit, Caglar Yildirim","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"训练生命周期的数据污染审计论文","best_for_zh":"维护推理基准、评估后训练模型或设计污染审计的研究者。","confidence":"medium","one_line":["Controlled clean-versus-contaminated models show that SFT and GRPO can resurface leakage after extended pre-training.","受控配对实验表明，SFT 与 GRPO 会重新暴露预训练污染，且外部迁移边界不同。"],"why":"It makes post-training a necessary stage in contamination audits rather than treating base-model measurements as definitive.","primary_link":"https://openreview.net/forum?id=GFDSGlEks2","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=GFDSGlEks2&name=supplementary_material"}],"link_count":2,"sections":9},{"id":"limits-inference-scaling-resampling-2026","title":"The Limits of Inference Scaling Through Resampling","year":2026,"venue":"ICLR 2026","authors":["Benedikt Stroebl","Sayash Kapoor","Arvind Narayanan"],"authors_zh":"Benedikt Stroebl、Sayash Kapoor、Arvind Narayanan","tracks":["data_construction_open_release_recipes"],"source_role":["scaling_study","audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","audit","test_time_compute"],"construction_layer":["reward_verifier_layer","scaling_report","release_audit"],"domains":["code"],"tags":["imperfect-verifier","rejection-sampling","verifier-filtering","false-positives","inference-scaling","code-reasoning","data-curation-audit","evalplus","optimal-stopping"],"status":"partial","priority":"可读","paper_type_zh":"不完美 verifier 与重采样上限研究","best_for_zh":"研究 rejection sampling、verifier 误接受、测试时计算扩展与推理数据审计的读者","confidence":"high","one_line":["Audits verifier-gated code resampling with weak and extended unit tests, showing that more attempts can preserve false positives and make a finite or zero budget optimal.","该研究用原始单元测试作为弱验收器、用 EvalPlus 扩展测试进行追溯审计，表明增加重采样次数仍会保留错误代码，并可能使有限预算乃至零预算最优。"],"why":"Rejection-sampled reasoning data are only as reliable as their acceptance gate; this work makes verifier precision, generator capability, task difficulty, and sampling budget explicit audit dimensions.","primary_link":"https://openreview.net/forum?id=j8H84v6AZ1","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/benediktstroebl/inference-scaling-limits"},{"key":"data","label":["Data","数据"],"url":"https://www.dropbox.com/scl/fi/je3d9lrmu36g5x3alugsa/humaneval_evalplus.zip?dl=0&rlkey=del4cd36kfyyseaw7r9zs8gn0&st=6ixn93qb"}],"link_count":6,"sections":9},{"id":"minimax-m2-2026","title":"The MiniMax-M2 Series: Mini Activations Unleashing Max Real-World Intelligence","year":2026,"venue":"arXiv preprint","authors":["MiniMax"],"authors_zh":"MiniMax","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["environmental"],"supervision_granularity":["full_episode"],"training_use":["agent_training","rlvr"],"construction_layer":["frontier_pipeline","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agentic_coding","agentic_cowork","software_environment"],"tags":["frontier-report","minimax","agentic-rl","executable-workspace","artifact-aligned-reward","forge","disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿智能体模型技术报告与数据披露账本","best_for_zh":"审计可执行工作区、智能体轨迹和长程 RL 披露边界的读者","confidence":"medium","one_line":["MiniMax-M2 reports executable-workspace trajectories and artifact-aligned rewards under Forge agent RL, but not the task corpus, validator implementation, or replayable trajectory ledger.","MiniMax-M2 报告了由可执行工作区、工件对齐奖励和 Forge 驱动的长程智能体 RL，但未公开任务、环境、奖励实现或轨迹台账。"],"why":"It is a clear case where environmental reward language is valuable evidence but not a substitute for auditable task, trajectory, and verifier artifacts.","primary_link":"https://arxiv.org/abs/2605.26494","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MiniMax-AI/MiniMax-M2"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/MiniMaxAI/MiniMax-M2"}],"link_count":3,"sections":9},{"id":"periodic-table-llm-reasoning-2026","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","year":2026,"venue":"arXiv preprint","authors":["Avinash Anand","Mahisha Ramesh","Avni Mittal","Ashutosh Kumar","Erik Cambria","Zhengkui Wang","Timothy Liu","Aik Beng Ng","Simon See","Rajiv Ratn Shah"],"authors_zh":"Avinash Anand 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":["llm-reasoning","reasoning-taxonomy","failure-modes","evaluation-methodology"],"tags":["reasoning-survey","taxonomy","failure-modes","evaluation",2026],"status":"verified","priority":"必读","paper_type_zh":"大语言模型推理综述","best_for_zh":"整理推理范式、方法、基准与失效模式的结构化综述。","confidence":"high","one_line":["A 2026 reasoning survey that maps paradigms, method families, benchmarks, and failure modes before readers enter data-specific tracks.","一篇把推理范式、方法趋势、评测与失效模式放入同一张表的 2026 综述。"],"why":"It makes the distinction between a reasoning task label and the evidence needed to train or evaluate it explicit.","primary_link":"https://arxiv.org/abs/2606.11470","links":[],"link_count":2,"sections":9},{"id":"signal-in-steps-lalp-2026","title":"The Signal is in the Steps: Local Scoring for Reasoning Data Selection","year":2026,"venue":"ICML 2026","authors":["Hoang Anh Just","Myeongseob Ko","Ruoxi Jia"],"authors_zh":"Hoang Anh Just、Myeongseob Ko、Ruoxi Jia","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","model_report","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["sft","distillation"],"construction_layer":["trace_writing","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning","scientific_reasoning","code_reasoning"],"tags":["lalp","reasoning-data-selection","local-log-probability","mixed-teacher-distillation","step-segmentation","student-specific-curation","response-selection","release-incomplete"],"status":"partial","priority":"可读","paper_type_zh":"推理数据选择 recipe 与全局似然失效分析","best_for_zh":"研究 reasoning distillation、mixed-teacher response selection、step-local likelihood 或数据选择审计的读者","confidence":"high","one_line":["The method turns step-local student likelihood into an offline selector for mixed-teacher reasoning traces, improving reported SFT results while releasing neither the selected corpus nor a currently usable implementation.","LALP 在 817 个 LIMO prompt 上用 GLM-4.5-Air 切分候选完整 response，再按目标 student 的局部 step likelihood 排序并保留每题一条完整轨迹；它是 student-specific 的离线数据选择器，不是 step correctness/process supervision，且当前匿名仓库没有可用实现或数据。"],"why":"It distinguishes useful local reasoning transitions from global fluency and self-copying, but also shows that a data-quality score can be student-specific, segmentation-sensitive, non-verifying, and difficult to audit without the candidate and score ledger.","primary_link":"https://arxiv.org/abs/2510.03988","links":[],"link_count":6,"sections":9},{"id":"throw-hybrid-test-time-compute-2026","title":"Think Hard Only When Needed: A Hybrid Best-of-N and Beam Search for Efficient Test-Time Compute","year":2026,"venue":"Findings of EACL 2026","authors":["Hyewon Suh","Chaojian Li","Cheng-Jhih Shih","Zheng Wang","Kejing Xia","Yonggan Fu","Yingyan Celine Lin"],"authors_zh":"Hyewon Suh、Chaojian Li、Cheng-Jhih Shih、Zheng Wang、Kejing Xia、Yonggan Fu、Yingyan Celine Lin（机构：佐治亚理工学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","hybrid-search","best-of-n","beam-search","process-reward-model"],"status":"verified","priority":"必读","paper_type_zh":"自适应混合测试时计算研究","best_for_zh":"服务批量大小为一的推理负载、且必须平衡并行采样吞吐与束搜索顺序评分成本的读者。","confidence":"high","one_line":["THROW uses early PRM scores to spend less search on easy prompts and selectively expand promising branches for hard ones.","THROW 利用早期过程奖励模型分数，让简单提示少花搜索预算，并为困难提示选择性扩展有前景的分支。"],"why":"It turns query difficulty into a concrete switching rule between two common search families instead of applying the same compute budget to every prompt.","primary_link":"https://aclanthology.org/2026.findings-eacl.315/","links":[],"link_count":2,"sections":9},{"id":"thinkbooster-2026","title":"ThinkBooster: A Unified Framework for Seamless Test-Time Scaling of LLM Reasoning","year":2026,"venue":"ACL 2026 System Demonstrations","authors":["Vladislav Smirnov","Quang-Chieu Nguyen","Sergey Senichev","Minh Ngoc Ta","Ekaterina Fadeeva","Artem Vazhentsev","Daria Galimzianova","Nikolai Rozanov","Viktor Mazanov","Jingwei Ni","Tianyi Wu","Igor Kiselev","Mrinmaya Sachan","Iryna Gurevych","Preslav Nakov","Timothy Baldwin","Artem Shelmanov"],"authors_zh":"Vladislav Smirnov、Quang-Chieu Nguyen、Sergey Senichev、Minh Ngoc Ta、Ekaterina Fadeeva、Artem Vazhentsev、Daria Galimzianova、Nikolai Rozanov、Viktor Mazanov、Jingwei Ni、Tianyi Wu、Igor Kiselev、Mrinmaya Sachan、Iryna Gurevych、Preslav Nakov、Timothy Baldwin、Artem Shelmanov","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["infrastructure","scaling_study","benchmark","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","full_episode","scalar_reward","process_reward"],"training_use":["evaluation","audit","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematical-reasoning","scientific-question-answering","code-generation","cuda-kernel-generation","test-time-compute","reasoning-search"],"tags":["test-time-compute","reasoning-search","best-of-n","self-consistency","beam-search","extended-thinking","mur","deepconf","phi-decoding","uncertainty-cot","process-reward-model","uncertainty-scoring","llm-critic","reprobe","visual-debugger","candidate-traces","selection-logs","compute-accounting","openai-compatible-api","vllm","infrastructure"],"status":"partial","priority":"必读","paper_type_zh":"测试时推理搜索基础设施、扩展研究与轨迹审计案例","best_for_zh":"研究测试时计算、候选轨迹、PRM 与不确定性选择器、搜索调试、推理预算归因和运行产物发布审计的读者","confidence":"high","one_line":["ThinkBooster is an MIT-licensed toolkit and ACL 2026 demo that generates and inspects prompt, candidate, step-score, selection, token, TFLOP, latency, and config traces for nine test-time reasoning strategies, but does not release the historical paper-run trace corpus.","ThinkBooster 是一个采用 MIT 许可证的测试时推理工具包，为九种策略统一记录候选轨迹、步骤分数、选择、token、理论 TFLOPs、延迟与配置；公开发布只有两个 Claude Sonnet 4 调试器示例，而没有论文运行的逐样本轨迹语料。"],"why":"It provides a common runtime data contract for comparing sampled trajectories, search trees, PRM, confidence, and critic selectors, and exact inference budgets, while making release completeness, scorer domain shift, step segmentation, systems accounting, API drift, and generated-output provenance auditable.","primary_link":"https://aclanthology.org/2026.acl-demo.70/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/IINemo/thinkbooster"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/test-time-compute"},{"key":"project","label":["Project","项目主页"],"url":"http://thinkbooster.nlpresearch.group"}],"link_count":7,"sections":9},{"id":"min-seek-sequential-tts-2026","title":"Thinking Long, but Short: Stable Sequential Test-Time Scaling for Large Reasoning Models","year":2026,"venue":"Findings of EACL 2026","authors":["Michael R. Metel","Yufei Cui","Boxing Chen","Prasanna Parthasarathi"],"authors_zh":"Michael R. Metel、Yufei Cui、Boxing Chen、Prasanna Parthasarathi（机构：华为诺亚方舟实验室）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","sequential-scaling","kv-cache","reasoning-length","efficiency"],"status":"verified","priority":"可读","paper_type_zh":"免训练的串行测试时扩展研究（Findings of EACL 2026）","best_for_zh":"研究长推理、推理稳定性与 KV 缓存高效扩展的读者。","confidence":"high","one_line":["Min-Seek stabilizes sequential test-time scaling by retaining only short reconstruction thoughts in a custom KV cache.","Min-Seek 通过在定制 KV 缓存中仅保留短重构思考，稳定串行测试时扩展。"],"why":"It treats memory retention as a first-class control over whether additional reasoning steps help or destabilize a model.","primary_link":"https://aclanthology.org/2026.findings-eacl.153/","links":[],"link_count":2,"sections":9},{"id":"ltpo-latent-thought-2026","title":"Thinking on the Fly: Test-Time Reasoning Enhancement via Latent Thought Policy Optimization","year":2026,"venue":"ICLR 2026","authors":["Wengao Ye","Yan Liang","Lianlei Shan"],"authors_zh":"Wengao Ye、Yan Liang、Lianlei Shan（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["Thinking on the Fly: Test-Time Reasoning Enhancement via Latent Thought Policy Optimization","潜在推理避免冗长文本链，却可能在最需要稳健测试时推理的困难或分布外问题上崩溃。"],"why":"It makes an inference-budget decision auditable.","primary_link":"https://arxiv.org/abs/2510.04182","links":[],"link_count":2,"sections":9},{"id":"thinking-free-policy-initialization-2025","title":"Thinking-Free Policy Initialization Makes Distilled Reasoning Models More Effective and Efficient Reasoners","year":2026,"venue":"ICLR 2026","authors":["Xin Xu","Clive Bai","Kai Yang","Tianhao Chen","Yang Wang","Saiyong Yang","Can Yang"],"authors_zh":"Xin Xu；Clive Bai；Kai Yang；Tianhao Chen；Yang Wang；Saiyong Yang；Can Yang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","distillation","evaluation"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["rlvr","thinking-free","policy-initialization","polaris53k","efficient-reasoning"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["An RLVR initialization recipe aimed at reducing explicit reasoning cost while retaining answer-level reward training.","TFPI 以空的 closed-think 前缀初始化策略，在 Polaris-53K 上使用 8 个 RLVR rollout 训练较少显式思考但保持性能的模型。"],"why":"It connects rollout-based policy training to an efficiency goal, but must be documented with verifier, prompt, and rollout lineage rather than only released model checkpoints.","primary_link":"https://arxiv.org/abs/2509.26226","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent-Hunyuan/Thinking-Free_Policy_Initialization"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/xx18/tfpi"}],"link_count":3,"sections":9},{"id":"programs-future-evaluation-2026","title":"Time to Impeach LLM-as-a-Judge: Programs are the Future of Evaluation","year":2026,"venue":"ICML 2025 Workshop on Programmatic Representations for Agent Learning","authors":["Tzu-Heng Huang","Harit Vishwakarma","Frederic Sala"],"authors_zh":"Tzu-Heng Huang、Harit Vishwakarma、Frederic Sala","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"可执行程序式 LLM judge、弱监督聚合与评测审计研究","best_for_zh":"需要低成本、可审计地重复执行固定 rubric 评测的研究者。","confidence":"medium","one_line":["Contrasts program-based judges with LLM judges for bias and consistency failures; status is submission.","PAJAMA 将 LLM 的判决逻辑合成为可执行程序，以降低成本并改善偏差与一致性。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://arxiv.org/abs/2506.10403","links":[],"link_count":3,"sections":9},{"id":"timely-machine-agentic-tts-2026","title":"Timely Machine: Awareness of Time Makes Test-Time Scaling Agentic","year":2026,"venue":"ACL 2026","authors":["Yichuan Ma","Linyang Li","Yongkang Chen","Peiji Li","Xiaozhe Li","Qipeng Guo","Dahua Lin","Kai Chen"],"authors_zh":"Yichuan Ma、Linyang Li、Yongkang Chen、Peiji Li、Xiaozhe Li、Qipeng Guo、Dahua Lin、Kai Chen（机构：上海人工智能实验室、复旦大学、香港中文大学、同济大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","sft","reinforcement_learning","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-reasoning","mathematical-reasoning","software-engineering","tool-use"],"tags":["test-time-scaling","agentic-reasoning","wall-clock-time","tool-latency","time-aware-rl"],"status":"verified","priority":"必读","paper_type_zh":"智能体测试时扩展基准与时间感知强化学习研究","best_for_zh":"工具延迟与模型生成速度共同决定额外推理能否满足真实截止时间的智能体开发者。","confidence":"high","one_line":["Timely Machine makes wall-clock time, rather than token length, the test-time budget for tool agents and trains policies to adapt their interaction strategy to latency.","Timely Machine 将工具智能体的测试时预算定义为墙钟时间而非词元长度，并训练策略依据延迟调整交互方式。"],"why":"It identifies wall-clock latency as a missing allocation variable and demonstrates that the best model size and interaction policy can reverse as tools become slower.","primary_link":"https://aclanthology.org/2026.acl-long.211/","links":[],"link_count":2,"sections":9},{"id":"tmas-multi-agent-synergy-2026","title":"TMAS: Scaling Test-Time Compute via Multi-Agent Synergy","year":2026,"venue":"arXiv preprint","authors":["George Wu","Nan Jing","Qing Yi","Chuan Hao","Ming Yang","Feng Chang","Yuan Wei","Jian Yang","Ran Tao","Bryan Dai"],"authors_zh":"George Wu、Nan Jing、Qing Yi、Chuan Hao、Ming Yang、Feng Chang、Yuan Wei、Ran Tao、Bryan Dai（IQuest Research）；Jian Yang（北京航空航天大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","multi-agent","memory","verification","rlvr"],"status":"verified","priority":"可读","paper_type_zh":"多智能体迭代测试时扩展研究","best_for_zh":"设计长程推理中探索、验证与记忆复用协同机制的读者。","confidence":"high","one_line":["TMAS coordinates parallel solution, verification, summarization, and two memory banks so later test-time iterations can reuse reliable progress without repeating strategies.","TMAS 让解答、验证、摘要与双层记忆智能体在多轮推理中协同，从而复用可靠进展并避免重复探索。"],"why":"It treats cross-trajectory information retention as a controllable test-time resource rather than leaving parallel rollouts independent.","primary_link":"https://arxiv.org/abs/2605.10344","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/IQuestLab/tmas"}],"link_count":3,"sections":9},{"id":"toss-token-level-safe-finetuning-2026","title":"Token-level Data Selection for Safe LLM Fine-tuning","year":2026,"venue":"ICLR 2026","authors":["Yanping Li","Zhening Liu","Zijian Li","Zehong Lin","Jun Zhang"],"authors_zh":"Yanping Li, Zhening Liu, Zijian Li, Zehong Lin, Jun Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","audit_failure"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["sft","safety_alignment"],"construction_layer":["reward_verifier_layer","optimizer_scaffold"],"domains":["safety-alignment","instruction-tuning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"必读","paper_type_zh":"安全微调的词元级数据选择论文（ICLR 2026）","best_for_zh":"研究安全微调、训练数据清洗和安全—效用权衡的读者。","confidence":"high","one_line":["TOSS masks unsafe response tokens before SFT using a loss-difference signal from safety and utility reference models.","TOSS 依据安全与效用参考模型的损失差，在 SFT 前掩掉不安全回复词元。"],"why":"It makes the connection between a data object and its training objective inspectable.","primary_link":"https://arxiv.org/abs/2603.01185","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Polly-LYP/TOSS"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Polly1231/TOSS"}],"link_count":5,"sections":9},{"id":"moral-roleplay-villains-2026","title":"Too Good to be Bad: On the Failure of LLMs to Role-Play Villains","year":2026,"venue":"Findings of ACL 2026","authors":["Zihao Yi","Qingxuan Jiang","Ruotian Ma","Xingyu Chen","Qu Yang","Mengru Wang","Fanghua Ye","Ying Shen","Zhaopeng Tu","Xiaolong Li","Liefeng Bo"],"authors_zh":"Zihao Yi、Qingxuan Jiang、Ruotian Ma、Xingyu Chen、Qu Yang、Mengru Wang、Fanghua Ye、Ying Shen、Zhaopeng Tu、Xiaolong Li、Liefeng Bo","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","answer_level"],"training_use":["evaluation","audit"],"construction_layer":["filtering_curation","trace_writing","release_audit"],"domains":["roleplay","safety_alignment","creative_writing"],"tags":["roleplay","safety","rubric","llm_judge"],"status":"verified","priority":"可读","paper_type_zh":"安全与角色一致性基准论文","best_for_zh":"研究安全对齐、角色扮演保真度或特质化 Judge 的研究者。","confidence":"high","one_line":["Moral RolePlay evaluates whether safety-aligned LLMs can preserve a fictional persona across a four-level moral spectrum.","Moral RolePlay 以四级道德谱系和角色特质标签，测量安全对齐模型能否忠实扮演虚构反派。"],"why":"It exposes a judgment-required failure where generic chat quality is not a valid substitute for character fidelity.","primary_link":"https://aclanthology.org/2026.findings-acl.282/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent/DigitalHuman/tree/main/RolePlay_Villain"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Zihao1/Moral-RolePlay"}],"link_count":4,"sections":9},{"id":"toolrm-agentic-tool-use-2026","title":"ToolRM: Towards Agentic Tool-Use Reward Modeling","year":2026,"venue":"Findings of ACL 2026","authors":["Renhao Li","Jianhong Tu","Yang Su","Yantao Liu","Fei Huang","Hamid Alinejad-Rokny","Derek F. Wong","Junyang Lin","Min Yang"],"authors_zh":"Renhao Li、Jianhong Tu、Yang Su、Yantao Liu、Fei Huang、Hamid Alinejad-Rokny、Derek F. Wong、Junyang Lin、Min Yang","tracks":["training_usage_optimization_objectives","preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["agentic-reasoning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"工具调用奖励模型与成对偏好轨迹论文","best_for_zh":"需要从工具轨迹构造可验证偏好并训练智能体奖励模型的读者。","confidence":"high","one_line":["ToolRM builds verifiable pairwise preferences from tool-use trajectories and trains generative and discriminative reward models for agentic optimization.","ToolRM 从工具使用轨迹构造可验证的成对偏好，并训练生成式和判别式奖励模型服务智能体优化。"],"why":"It makes the connection between a data object and a training objective inspectable.","primary_link":"https://aclanthology.org/2026.findings-acl.419/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lirenhao1997/ToolRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/RioLee/ToolPref-Pairwise-30K"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/RioLee/toolrm"}],"link_count":5,"sections":9},{"id":"mechanistic-lrm-survey-2026","title":"Towards a Mechanistic Understanding of Large Reasoning Models: A Survey of Training, Inference, and Failures","year":2026,"venue":"ACL 2026 Long Papers","authors":["Yi Hu","Jiaqi Gu","Ruxin Wang","Zijun Yao","Hao Peng","Xiaobao Wu","Jianhui Chen","Muhan Zhang","Liangming Pan"],"authors_zh":"Yi Hu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","large-reasoning-models","mechanistic-interpretability","reasoning-data"],"tags":["foundations-and-primers","large-reasoning-models","interpretability","acl-2026","survey"],"status":"verified","priority":"必读","paper_type_zh":"大推理模型机制综述","best_for_zh":"研究推理过程、强化学习或模型行为证据的读者。","confidence":"high","one_line":["An ACL survey of what evidence explains how large reasoning models learn, reason, and fail.","从训练、推理过程与失效行为三条线理解大推理模型的 ACL 综述。"],"why":"It helps readers distinguish an impressive answer from evidence about the process that produced it.","primary_link":"https://aclanthology.org/2026.acl-long.889/","links":[],"link_count":2,"sections":9},{"id":"hierarchical-multistep-reward-models-2026","title":"Towards Hierarchical Multi-Step Reward Models for Enhanced Reasoning in Large Language Models","year":2026,"venue":"Findings of ACL 2026","authors":["Teng Wang","Jiang Zhangyi","Zhenqi He","Hailei Gong","Shenyang Tong","Wenhan Yang","Zeyu Li","Yanan Zheng","Zifan He","Zewen Ye","Shengjie Ma","Jianping Zhang"],"authors_zh":"Teng Wang、Jiang Zhangyi、Zhenqi He、Hailei Gong、Shenyang Tong、Wenhan Yang、Zeyu Li、Yanan Zheng、Zifan He、Zewen Ye、Shengjie Ma、Jianping Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["process_reward"],"training_use":["process_supervision"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning"],"tags":["post-training","training-usage"],"status":"verified","priority":"可读","paper_type_zh":"过程奖励模型与多步监督论文","best_for_zh":"研究 PRM 训练样本、步骤标签与推理过程奖励的读者。","confidence":"high","one_line":["HRM scores individual and consecutive reasoning steps and augments process-supervision data through hierarchical node compression.","HRM 同时判断单个步骤和连续步骤，并以节点压缩扩充过程监督数据来训练更稳健的奖励模型。"],"why":"It makes the link between an explicit data object and a training objective inspectable.","primary_link":"https://aclanthology.org/2026.findings-acl.27/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tengwang0318/hierarchial_reward_model"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LLM4OR/StructuredOR"}],"link_count":5,"sections":9},{"id":"continuous-space-inference-scaling-2026","title":"Towards Inference-time Scaling for Continuous Space Reasoning","year":2026,"venue":"Findings of ACL 2026","authors":["Minghan Wang","Thuy-Trang Vu","Ehsan Shareghi","Gholamreza Haffari"],"authors_zh":"Minghan Wang、Thuy-Trang Vu、Gholamreza Haffari（莫纳什大学）；Ehsan Shareghi（伦敦大学学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","reinforcement_learning","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["inference-time-scaling","latent-reasoning","reward-model","reranking","continuous-space"],"status":"verified","priority":"可读","paper_type_zh":"连续隐空间测试时扩展分析","best_for_zh":"希望把采样与验证器扩展从文本思维链迁移到潜在或连续推理表征的读者。","confidence":"high","one_line":["This paper shows that latent reasoning has unused Pass@N potential but weak reward-model separability, then tests geometric regularization to improve its test-time scaling.","本文显示隐空间推理虽有尚未利用的通过率上界，却因奖励模型可分性弱而难以扩展，并以几何正则化测试改进测试时扩展的可能。"],"why":"It establishes that standard verifier recipes cannot simply be assumed to transfer from text traces to continuous thoughts, a key boundary for generalizing test-time scaling.","primary_link":"https://aclanthology.org/2026.findings-acl.1338/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yuriak/LatentITS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/peiyi9979/Math-Shepherd"}],"link_count":4,"sections":9},{"id":"tif-lossdiff-irm-preference-selection-2026","title":"Towards Understanding Valuable Preference Data for Large Language Model Alignment","year":2026,"venue":"ICLR 2026","authors":["Zizhuo Zhang","Qizhou Wang","Shanshan Ye","Jianing Zhu","Jiangchao Yao","Bo Han","Masashi Sugiyama"],"authors_zh":"Zizhuo Zhang、Qizhou Wang、Shanshan Ye、Jianing Zhu、Jiangchao Yao、Bo Han、Masashi Sugiyama（香港浸会大学、悉尼科技大学、上海交通大学、东京大学、RIKEN AIP）","tracks":["training_usage_optimization_objectives"],"source_role":["scaling_study","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-following"],"tags":["preference-data","data-selection","dpo","influence"],"status":"verified","priority":"必读","paper_type_zh":"模型相对的偏好数据选择与对齐研究","best_for_zh":"适合需要用较少偏好对完成 DPO 或 SLiC 对齐的读者。","confidence":"high","one_line":["LossDiff-IRM selects preference pairs by approximating their model-specific validation influence.","该方法用近似验证集影响的 LossDiff-IRM 选择对当前模型真正有益的偏好对。"],"why":"It shows that a preference pair's value is a model-relative training property, not an intrinsic label.","primary_link":"https://arxiv.org/abs/2510.13212","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tmlr-group/TIF_LossDiff-IRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuggingFaceH4/ultrafeedback_binarized"}],"link_count":5,"sections":9},{"id":"tracing-the-roots-2026","title":"Tracing the Roots: A Multi-Agent Framework for Uncovering Data Lineage in Post-Training LLMs","year":2026,"venue":"ACL 2026","authors":["Yu Li","Xiaoran Shang","Qizhi Pei","Yun Zhu","Xin Gao","Honglin Lin","Zhanping Zhong","Zhuoshi Pan","Zheng Liu","Xiaoyang Wang","Conghui He","Dahua Lin","Feng Zhao","Lijun Wu"],"authors_zh":"Yu Li、Xiaoran Shang、Qizhi Pei、Yun Zhu、Xin Gao、Honglin Lin、Zhanping Zhong、Zhuoshi Pan、Zheng Liu、Xiaoyang Wang、Conghui He、Dahua Lin、Feng Zhao、Lijun Wu","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","audit_failure","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","code","science"],"tags":["data-lineage","provenance","multi-agent-curation","decontamination","root-source-sampling","dataset-diversity","release-versioning"],"status":"partial","priority":"可读","paper_type_zh":"数据谱系审计与构造方法论文","best_for_zh":"关注后训练数据来源、混合去重、污染传播、根源采样与发布可审计性的研究者","confidence":"medium","one_line":["Reconstructs a 430-node, 971-edge post-training dataset lineage graph and uses 31 upstream QA sources plus exact/MinHash deduplication to report a 570K diversity-oriented corpus that is not included in the verified release.","重建含 430 个节点、971 条边的后训练数据集谱系图，并从 31 个上游问答源经精确匹配与 MinHash 去重报告 57 万条多样性导向语料；经核验的官方发布并不包含该语料。"],"why":"For the construction/open-release track, it turns upstream ancestry into a selection and audit layer, while showing why dataset-level edges, record-level provenance, benchmark overlap, and release licensing must be checked separately.","primary_link":"https://arxiv.org/abs/2604.10480","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Leey21/data-lineage"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Leey21/data-lineage/tree/main/output"},{"key":"project","label":["Project","项目主页"],"url":"https://arena.opendatalab.org.cn/data-lineage/website/index.html"}],"link_count":7,"sections":9},{"id":"tree-search-agent-rl-2026","title":"Tree Search for LLM Agent Reinforcement Learning","year":2026,"venue":"ICLR 2026","authors":["Yuxiang Ji","Ziyu Ma","Yong Wang","Guanhua Chen","Xiangxiang Chu","Liaoni Wu"],"authors_zh":"Yuxiang Ji、Ziyu Ma、Yong Wang、Guanhua Chen、Xiangxiang Chu、Liaoni Wu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","process_supervision","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["knowledge_intensive_question_answering","multi_hop_question_answering","web_agent_question_answering","retrieval_augmented_reasoning","tool_use","agentic_reasoning"],"tags":["tree-grpo","agent-step-tree-search","shared-prefix-rollouts","agentic-rl","react","online-rlvr","outcome-reward","implicit-process-supervision","intra-tree-advantage","inter-tree-advantage","retrieval-agent","rollout-budget","unreleased-raw-trees"],"status":"partial","priority":"必读","paper_type_zh":"在线 agent RL 树式 rollout 构造、隐式过程监督与发布审计研究","best_for_zh":"研究共享前缀 rollout、ReAct agent-step 信用分配、训练预算、程序化奖励和轨迹发布边界的读者","confidence":"high","one_line":["Tree-GRPO builds shared-prefix forests of complete ReAct agent steps during online RL, scores terminal leaves with EM/F1 and format rules, and derives intra-/inter-tree process signals, but it releases code rather than the raw trees, discarded branches, or reward lineage.","Tree-GRPO 在在线 RL 中构造共享前缀的 ReAct agent-step 树，以 EM/F1 与格式规则给终点叶子评分并形成树内/树间相对优势；官方发布提供实现代码，但未发布原始树、舍弃分支或逐记录奖励谱系。"],"why":"It shows how search structure itself can turn a fixed token/tool budget into more agent trajectories and finer credit assignment, while exposing why parent-child lineage, all leaf outcomes, environment snapshots, split manifests, and retention decisions are necessary before such runtime trees can be treated as auditable reasoning data.","primary_link":"https://openreview.net/forum?id=ZpQwAFhU13","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AMAP-ML/Tree-GRPO"}],"link_count":6,"sections":9},{"id":"treepo-2026","title":"TreePO: Enhancing Policy Efficacy and Inference Efficiency with Tree Modeling","year":2026,"venue":"ICML 2026","authors":["Yizhi LI","Qingshui Gu","Zhoufutu Wen","Ziniu Li","Ruibin Yuan","Tianshun Xing","Shuyue Guo","Tuney Zheng","Xin Zhou","Xingwei Qu","Wangchunshu Zhou","Zheng Zhang","Wei Shen","Wei Xue","Qian Liu","Chenghua Lin","Jian Yang","Ge Zhang","Wenhao Huang"],"authors_zh":"Yizhi LI、Qingshui Gu、Zhoufutu Wen、Ziniu Li、Ruibin Yuan、Tianshun Xing、Shuyue Guo、Tuney Zheng、Xin Zhou、Xingwei Qu、Wangchunshu Zhou、Zheng Zhang、Wei Shen、Wei Xue、Qian Liu、Chenghua Lin、Jian Yang、Ge Zhang、Wenhao Huang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward","trajectory_value"],"training_use":["rlvr","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematical_reasoning","reasoning"],"tags":["treepo","tree-structured-rollouts","shared-prefix","segment-level-sampling","kv-cache-reuse","hierarchical-advantage","outcome-reward","rlvr","test-time-compute","inference-efficiency"],"status":"partial","priority":"必读","paper_type_zh":"树结构在线RLVR与推理效率研究","best_for_zh":"研究推理rollout数据、共享前缀搜索、RLVR信用分配、测试时计算与可复现性审计的读者","confidence":"high","one_line":["TreePO replaces independent math rollouts with shared-prefix segment trees and trains on terminal answer rewards through hierarchical subgroup advantages, releasing code, prompts, and checkpoints but not complete paper-run trees.","TreePO以共享前缀分段树替代独立数学rollout，并用终局答案奖励构造层级子组优势；官方发布了代码、提示与检查点，但没有发布可重建论文训练过程的完整树记录。"],"why":"It exposes how rollout topology, branch allocation, fallback, token budgets, and outcome-reward credit assignment jointly shape RLVR efficiency, while showing why prompt releases alone are insufficient for branch-level reproducibility.","primary_link":"https://openreview.net/forum?id=npsWK8rgYO","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/multimodal-art-projection/TreePO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/m-a-p/TreePO_data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/m-a-p/treepo"},{"key":"project","label":["Project","项目主页"],"url":"https://m-a-p.ai/TreePO/"}],"link_count":8,"sections":9},{"id":"trimr-thinking-trimming-2026","title":"TrimR: Verifier-based Training-Free Thinking Trimming for Efficient Test-Time Scaling","year":2026,"venue":"ICLR 2026","authors":["Weizhe Lin","Xing Li","Zhiyuan Yang","Xiaojin Fu","Hui-Ling Zhen","Yaoyuan Wang","Xianzhi Yu","Wulong Liu","Xiaosong Li","Mingxuan Yuan"],"authors_zh":"Weizhe Lin、Xing Li、Zhiyuan Yang、Xiaojin Fu、Hui-Ling Zhen、Yaoyuan Wang、Xianzhi Yu、Wulong Liu、Xiaosong Li、Mingxuan Yuan（机构：华为先进计算与存储实验室、华为诺亚方舟实验室）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-compute","reasoning-compression","verifier","early-exit","serving"],"status":"verified","priority":"必读","paper_type_zh":"测试时推理效率与在线服务系统研究","best_for_zh":"研究验证器引导压缩与自适应停止的推理模型读者。","confidence":"high","one_line":["TrimR uses a lightweight verifier to trim redundant reasoning online and preserve computation for unresolved traces.","TrimR 用轻量验证器在线识别冗余推理，并把计算保留给尚未解决的推理轨迹。"],"why":"It demonstrates a deployed, training-free route from intermediate thought checks to lower test-time cost.","primary_link":"https://iclr.cc/virtual/2026/poster/10007390","links":[],"link_count":2,"sections":9},{"id":"trustjudge-2026","title":"TrustJudge: Inconsistencies of LLM-as-a-Judge and How to Alleviate Them","year":2026,"venue":"ICLR 2026","authors":["Yidong Wang","Yunze Song","Tingyuan Zhu","Xuanwang Zhang","Zhuohao Yu","Hao Chen","Chiyu Song","Qiufeng Wang","Zhen Wu","Xinyu Dai","Yue Zhang","Cunxiang Wang","Wei Ye","Shikun Zhang"],"authors_zh":"Yidong Wang，Yunze Song，Tingyuan Zhu，Xuanwang Zhang，Zhuohao Yu，Hao Chen，Chiyu Song，Qiufeng Wang，Zhen Wu，Xinyu Dai，Yue Zhang，Cunxiang Wang，Wei Ye，Shikun Zhang。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"LLM-as-a-Judge 一致性与可靠性审计","best_for_zh":"构建、选择或审计 LLM-as-a-Judge 自动评测系统的研究者与工程团队。","confidence":"medium","one_line":["Accepted study of judge inconsistency with public conference materials.","审计 LLM judge 的打分—比较冲突与成对传递性失效，并以概率推理降低不一致。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://openreview.net/forum?id=UwkNZPZ9Rl","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TrustJudge/TrustJudge"}],"link_count":4,"sections":9},{"id":"llm-oasis-end-to-end-factuality-evaluation-2026","title":"Truth or Mirage? Towards End-To-End Factuality Evaluation with LLM-Oasis","year":2026,"venue":"Computational Linguistics 2026","authors":["Alessandro Scir?","Andrei Stefan Bejgu","Simone Tedeschi","Karim Ghonim","Federico Martelli","Roberto Navigli"],"authors_zh":"Alessandro Scir?、Andrei Stefan Bejgu、Simone Tedeschi、Karim Ghonim、Federico Martelli、Roberto Navigli","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["evaluation","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","summarization"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"可读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["LLM-Oasis is the paper's released feedback or evaluation resource.","从维基事实构造真伪长文本对，再以人工验证和金标准测试集标注端到端事实性，关注生成文本整体而非孤立 claim 分类。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2026.cl-1.1/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/Babelscape/llm-oasis"}],"link_count":3,"sections":9},{"id":"tua-bench-2026","title":"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents","year":2026,"venue":"arXiv preprint","authors":["Shoufa Chen","Luyuan Wang","Xuan Yang","Zhiheng Liu","Yuren Cong","Yuanfeng Ji","Feiyan Zhou","Xiaohui Zhang","Fanny Yang","Belinda Zeng"],"authors_zh":"Shoufa Chen 等（Meta AI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["terminal-agents","general-purpose-terminal-use","execution-feedback"],"tags":["agent_environment","trajectory_data","terminal-agents","general-purpose-terminal-use","execution-feedback"],"status":"verified","priority":"可读","paper_type_zh":"arXiv 的 terminal-use agent benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["TUA-Bench benchmarks general-purpose terminal-use agents with executable tasks.","TUA-Bench 用可执行任务评测通用终端使用智能体。"],"why":"it expands terminal agents beyond coding and sysadmin tasks","primary_link":"https://arxiv.org/abs/2606.28480","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/TUA-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://tuabench.ai/"}],"link_count":4,"sections":9},{"id":"tumix-tool-use-mixture-2026","title":"TUMIX: Multi-Agent Test-Time Scaling with Tool-Use Mixture","year":2026,"venue":"ICLR 2026","authors":["Yongchao Chen","Jiefeng Chen","Rui Meng","Ji Yin","Na Li","Chuchu Fan","Chi Wang","Tomas Pfister","Jinsung Yoon"],"authors_zh":"Yongchao Chen、Jiefeng Chen、Rui Meng、Ji Yin、Na Li、Chuchu Fan、Chi Wang、Tomas Pfister、Jinsung Yoon（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning","software-engineering"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["TUMIX: Multi-Agent Test-Time Scaling with Tool-Use Mixture","没有一种工具使用策略适合每个问题，但代理常在测试时承诺单一路径的推理、编程或搜索。"],"why":"It makes a test-time decision auditable.","primary_link":"https://arxiv.org/abs/2510.01279","links":[],"link_count":2,"sections":9},{"id":"unprm-uncertainty-process-reward-data-2026","title":"Uncertainty-Based Methods for Automated Process Reward Data Construction and Output Aggregation in Mathematical Reasoning","year":2026,"venue":"AAAI 2026","authors":["Jiuzhou Han","Wray Buntine","Ehsan Shareghi"],"authors_zh":"Jiuzhou Han、Wray Buntine、Ehsan Shareghi","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-reward-modeling","uncertainty-estimation"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要用不确定性信号自动构造数学过程奖励数据，并比较过程奖励与答案频次聚合的研究者。","confidence":"high","one_line":["UnPRM uses uncertainty-guided sampling and automated step annotation to build mathematical process-reward data, then combines reward scores and answer frequencies for output aggregation.","UnPRM 发布约 4 万条数学过程样本，以不确定性筛选和自动验证赋予步骤正确性标签，直接服务 PRM 训练与答案聚合。"],"why":"It couples a scalable recipe for process-reward data construction with the downstream answer-aggregation rule that consumes those rewards.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40351","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Jiuzhouh/UnPRM"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Jiuzhouh/UnPRM/blob/main/data.zip"}],"link_count":5,"sections":9},{"id":"training-data-test-time-scaling-2026","title":"Understanding the Role of Training Data in Test-Time Scaling","year":2026,"venue":"ICLR 2026","authors":["Adel Javanmard","Baharan Mirzasoleiman","Vahab Mirrokni"],"authors_zh":"Adel Javanmard、Baharan Mirzasoleiman、Vahab Mirrokni","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","scaling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配的读者。","confidence":"high","one_line":["Understanding the Role of Training Data in Test-Time Scaling","更多思考无法修复训练数据从未提供的下游技能。"],"why":"It measures a concrete decision about additional inference computation.","primary_link":"https://arxiv.org/abs/2510.03605","links":[],"link_count":3,"sections":9},{"id":"unified-data-selection-2026","title":"Unified Data Selection for LLM Reasoning","year":2026,"venue":"ICLR 2026 submission / arXiv preprint","authors":["Xiaoyuan Li","Yubo Ma","Chengpeng Li","Fengbin Zhu","Yiyao Yu","Keqin Bao","Wenjie Wang","Fuli Feng","Dayiheng Liu"],"authors_zh":"Xiaoyuan Li、Yubo Ma、Chengpeng Li、Fengbin Zhu、Yiyao Yu、Keqin Bao、Wenjie Wang、Fuli Feng、Dayiheng Liu","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","trajectory_value"],"training_use":["sft","rlvr"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematical_reasoning","code_generation","stem_reasoning"],"tags":["high-entropy-tokens","data-selection","rejection-finetuning","supervised-finetuning","grpo","rollout-selection","correctness-filtering","length-sensitivity","tokenizer-sensitivity","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"跨阶段高熵数据选择研究","best_for_zh":"研究SFT、RFT、GRPO数据效率和不确定性筛选的读者","confidence":"medium","one_line":["Unified Data Selection ranks reasoning trajectories by entropy concentrated at their highest-uncertainty tokens and applies the score to SFT, correctness-filtered RFT and online GRPO rollout selection.","Unified Data Selection以回答内最高熵0.5%词元的熵和排序轨迹，并将其用于SFT、正确性过滤RFT和在线GRPO选择。"],"why":"It exposes a unified selection signal across three post-training stages and provides useful threshold, length and proxy-model ablations, but reuse still depends on undisclosed tokenizer details, correctness checks, record manifests and selected-data releases.","primary_link":"https://arxiv.org/abs/2605.22389","links":[],"link_count":4,"sections":9},{"id":"unirrm-unified-reasoning-reward-models-across-languages-and-evaluation-paradigms-2026","title":"UniRRM: Unified Reasoning Reward Models Across Languages and Evaluation Paradigms","year":2026,"venue":"ICML 2026","authors":["Peng Lai","Yichao Du","Junchao Wu","Weibo Gao","Linan Yue","Longyue Wang","Weihua Luo","Derek F. Wong","Guanhua Chen"],"authors_zh":"Peng Lai、Yichao Du、Junchao Wu、Weibo Gao、Linan Yue 等","tracks":["preference_reward_feedback_data","training_usage_optimization_objectives"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","UniRRM 以多语种混合奖励数据训练一个能处理成对、列表和单点评审的推理型奖励模型。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://openreview.net/forum?id=laiK6TlhL2","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SUSTech-NLP/UniRRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SUSTech-NLP/MixReward"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/SUSTech-NLP/UniRRM-14B"}],"link_count":4,"sections":9},{"id":"unisrm-2026","title":"UniSRM: A Unified Speech Reward Model for Reasoning-Based Fine-grained Assessment","year":2026,"venue":"ACL 2026","authors":["Yuanyuan Wang","Dongchao Yang","Yayue Deng","Zhiyong Wu","Yiwen Guo","Helen Meng","Xixin Wu"],"authors_zh":"Yuanyuan Wang、Dongchao Yang、Yayue Deng、Zhiyong Wu、Yiwen Guo、Helen Meng、Xixin Wu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"语音奖励模型、数据集与评测基准论文","best_for_zh":"需要构建带上下文的语音评委、语音奖励或审计多维判断记录的研究者。","confidence":"medium","one_line":["A speech reward-model dataset and benchmark for multi-dimensional, reasoning-based assessment.","以统一数据、逐维理由和一致性奖励，把语音质量判断从黑盒标量扩展为可审计的多任务奖励模型。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2026.acl-long.2150/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lavendery/UniSRM"}],"link_count":2,"sections":9},{"id":"v-droid-2025","title":"V-Droid: Advancing Mobile GUI Agent Through Generative Verifiers","year":2026,"venue":"ACM MobiCom 2026","authors":["Gaole Dai","Shiqi Jiang","Ting Cao","Yuanchun Li","Yuqing Yang","Rui Tan","Mo Li","Lili Qiu"],"authors_zh":"Gaole Dai、Shiqi Jiang、Ting Cao、Yuanchun Li、Yuqing Yang、Rui Tan、Mo Li、Lili Qiu","tracks":["environment_agent_trajectory_data"],"source_role":["model_report","verifier_reward","process_supervision","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","pairwise_preference","scalar_reward"],"training_use":["preference_learning","reward_modeling","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","mobile_gui_control","android","action_verification"],"tags":["v-droid","mobile-gui-agent","android","agent-trajectory","generative-verifier","pairwise-process-preference","process-supervision","human-agent-annotation","self-correction","release-incomplete","version-drift"],"status":"partial","priority":"可读","paper_type_zh":"移动 GUI 智能体的生成式动作验证器与偏好数据构建方法","best_for_zh":"研究移动智能体轨迹、过程偏好、验证器训练及发布审计的读者","confidence":"medium","one_line":["V-Droid trains a Llama-3.1-8B action verifier on 110K human-agent-annotated P3 preference pairs derived from Android trajectories, using entropy to target corrections and a small reverse-action subset for self-correction; the full corpus, replay state, licenses, and exact experiment-to-artifact mapping remain unavailable.","V-Droid 将 Android 轨迹扩展为 110K 组逐步 P3 动作偏好并训练标量验证器，但完整语料、可回放环境及实验—发布版本映射均未公开。"],"why":"It makes the feedback interface for mobile reasoning explicit - task state plus history and a candidate action maps to a learned scalar score - while showing how alternative-action labeling, uncertainty triage, environment replay, external memory models, and release/version drift determine whether an agent trajectory recipe can actually be reused.","primary_link":"https://doi.org/10.1145/3795866.3796681","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/V-Droid-Agent/V-Droid"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/V-Droid/V-Droid-8B-0323"},{"key":"project","label":["Project","项目主页"],"url":"https://v-droid-agent.github.io/"}],"link_count":6,"sections":9},{"id":"variation-in-verification-2025","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","year":2026,"venue":"ICLR 2026","authors":["Yefan Zhou","Austin Xu","Yilun Zhou","Janvijay Singh","Jiang Gui","Shafiq Joty"],"authors_zh":"Yefan Zhou, Austin Xu, Yilun Zhou, Janvijay Singh, Jiang Gui, Shafiq Joty","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematics","knowledge","natural_language_reasoning"],"tags":["generative-verifier","verification-cot","test-time-scaling","candidate-filtering","tpr-tnr","verifier-audit","open-traces"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"high","one_line":["A released 15-generator by 15-verifier trace matrix for auditing how reference-free generative verification changes with task difficulty and model capability.","Variation in Verification 系统审计生成式验证器在任务难度、生成器和验证器能力变化下的选择行为。"],"why":"It gives Track 5 a concrete, joinable record of candidate rollouts, correctness labels, verifier rationales, verdicts, and filtering behavior while exposing where test-time gains can be confounded by pool construction and verifier error.","primary_link":"https://arxiv.org/abs/2509.17995","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YefanZhou/llm-verify-dynamics"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/YefanZhou98/llm-verification-dynamics"},{"key":"project","label":["Project","项目主页"],"url":"https://yefanzhou.github.io/llm-verify-dynamic/"}],"link_count":6,"sections":9},{"id":"vericontest-verifiable-competitive-programming-2026","title":"VeriContest: A Competitive-Programming Benchmark for Verifiable Code Generation","year":2026,"venue":"arXiv","authors":["Zichen Xie","Mrigank Pawagi","Yuxin Liu","Aaditi Rai","Lize Shao","John Berberian Jr.","Sicong Che","Wenxi Wang"],"authors_zh":"Zichen Xie, Mrigank Pawagi, Yuxin Liu, Aaditi Rai, Lize Shao, John Berberian Jr., Sicong Che, Wenxi Wang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["formal-verification","rust","verus","competitive-programming","2026"],"status":"verified","priority":"可读","paper_type_zh":"Rust/Verus 形式化可验证代码生成 benchmark","best_for_zh":"需要 Verus 证明、规格生成或可验证程序合成评测数据的研究者。","confidence":"high","one_line":["VeriContest pairs 946 programming problems with Rust code, Verus specifications and proofs, and positive/negative tests to evaluate end-to-end verifiable code generation.","VeriContest 为 946 道竞赛编程题配套 Rust 代码、Verus 规格与证明以及正负测试，用于评测端到端可验证代码生成。"],"why":"It separates ordinary code generation from specification and proof synthesis while making all stages checkable by a formal verifier.","primary_link":"https://arxiv.org/abs/2605.08553","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HIPREL-Group/VeriContest"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Gax-c/VeriContest"}],"link_count":4,"sections":9},{"id":"verifiable-rewards-beyond-math-code-corpus-grounded-process-supervision-2026","title":"Verifiable Rewards Beyond Math and Code: Lightweight Corpus-Grounded Process Supervision for Factual Question Answering","year":2026,"venue":"arXiv","authors":["Shicheng Fan","Haochang Hao","Dehai Min","Weihao Liu","Philip S. Yu","Lu Cheng"],"authors_zh":"Shicheng Fan、Haochang Hao、Dehai Min、Weihao Liu、Philip S. Yu、Lu Cheng","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["factual_question_answering","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"面向事实问答的语料库验证过程奖励与训练数据论文","best_for_zh":"研究低成本事实性过程监督、句子级归因和强化学习奖励设计。","confidence":"high","one_line":["CorVer supplies corpus-grounded sentence-level rewards and 62.5K factual-QA training records without relying on a neural verifier in the reward loop.","CorVer 提供基于语料库的句子级奖励及 6.25 万条事实问答训练记录，无需在奖励环中调用神经验证器。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2605.29648","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Shichengf/CorVer-training-data"}],"link_count":2,"sections":9},{"id":"crv-verifying-cot-computational-graph-2026","title":"Verifying Chain-of-Thought Reasoning via Its Computational Graph","year":2026,"venue":"ICLR 2026 Oral","authors":["Zheng Zhao","Yeskendir Koishekenov","Xianjun Yang","Naila Murray","Nicola Cancedda"],"authors_zh":"Zheng Zhao、Yeskendir Koishekenov、Xianjun Yang、Naila Murray、Nicola Cancedda","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["chain-of-thought-verification","mechanistic-interpretability","mathematical-reasoning"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要白盒过程验证数据、带步骤正确性标签的 CoT，或希望把机制可解释性用于推理错误诊断的研究者。","confidence":"high","one_line":["CRV treats a reasoning step's attribution graph as an execution trace, using its structural features to identify and causally repair chain-of-thought errors.","CRV 把推理步骤的归因图视为计算执行轨迹，从图结构中识别并干预导致 CoT 错误的因果特征。"],"why":"It releases step-labeled traces and an explicit computational substrate, making verifier evidence traceable to internal graph structure rather than text alone.","primary_link":"https://openreview.net/pdf/df99704be0e2d31288aa12339efb101c39bcbb64.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/CRV"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/crv"}],"link_count":4,"sections":9},{"id":"verina-benchmarking-verifiable-code-generation","title":"VERINA: Benchmarking Verifiable Code Generation","year":2026,"venue":"ICLR 2026","authors":["Zhe Ye","Zhengxu Yan","Jingxuan He","Timothe Kasriel","Kaiyu Yang","Dawn Song"],"authors_zh":"Zhe Ye, Zhengxu Yan, Jingxuan He, Timothe Kasriel, Kaiyu Yang, Dawn Song","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** 189 Lean tasks provide code, specifications, proofs, and tests for modular and end-to-end vericoding.","189 个 Lean 任务同时提供代码、规格、证明和测试，支持模块化与端到端 vericoding。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://openreview.net/forum?id=0A4Uf88pog","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sunblaze-ucb/verina"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sunblaze-ucb/verina"},{"key":"project","label":["Project","项目主页"],"url":"https://verina.io/"}],"link_count":6,"sections":9},{"id":"video-based-reward-modeling-computer-use-agents-2026","title":"Video-Based Reward Modeling for Computer-Use Agents","year":2026,"venue":"ECCV 2026","authors":["Linxin Song","Jieyu Zhang","Huanxin Sheng","Taiwei Shi","Gupta Rahul","Yang Liu","Ranjay Krishna","Jian Kang","Jieyu Zhao"],"authors_zh":"Linxin Song、Jieyu Zhang、Huanxin Sheng 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["computer_use_agents","video_reward_modeling","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"基于执行视频的计算机使用智能体奖励建模与过程监督数据论文","best_for_zh":"训练或评测无需访问智能体内部动作与思维的轨迹成功判定器。","confidence":"high","one_line":["ExeVR-53k provides 53K instruction–execution-video–reward triplets and adversarial step annotations for judging computer-use trajectories from observable video.","ExeVR-53k 提供 5.3 万个指令—执行视频—奖励三元组及对抗式步骤标注，用可观测视频评判计算机使用轨迹。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2603.10178","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/limenlp/ExeVRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lime-nlp/ExeVR-53K"},{"key":"project","label":["Project","项目主页"],"url":"https://taiweis.com/papers/exevrm"}],"link_count":4,"sections":9},{"id":"video-mmmu-2025","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multidisciplinary Professional Videos","year":2026,"venue":"ACL 2026 main / arXiv","authors":["Kairui Hu","Penghao Wu","Fanyi Pu","Wang Xiao","Xiang Yue","Bo Li","Yuanhan Zhang","Ziwei Liu"],"authors_zh":"Kairui Hu 等（S-Lab, Nanyang Technological University、Carnegie Mellon University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["video-reasoning","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","video-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"ACL 2026 main / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["Video-MMMU exposes expert videos with perception, comprehension, and adaptation QA as an auditable evaluation surface.","Video-MMMU 把专家视频上的感知、理解与迁移应用问答做成可审计的评测面。"],"why":"Good lead for domain video reasoning rather than short clip recognition.","primary_link":"https://arxiv.org/abs/2501.13826","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EvolvingLMMs-Lab/VideoMMMU"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lmms-lab/VideoMMMU"},{"key":"project","label":["Project","项目主页"],"url":"https://videommmu.github.io/"}],"link_count":6,"sections":9},{"id":"visreason-visual-chain-of-thought-2026","title":"VisReason: A Large-Scale Dataset for Visual Chain-of-Thought Reasoning","year":2026,"venue":"ECCV 2026","authors":["Lingxiao Li","Yifan Wang","Xinyan Gao","Chen Tang","Xiangyu Yue","Chenyu You"],"authors_zh":"Lingxiao Li、Yifan Wang、Xinyan Gao、Chen Tang、Xiangyu Yue、Chenyu You","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["visual-chain-of-thought","multimodal-reasoning","spatial-grounding"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要区域级视觉过程监督、可核验视觉 CoT 或多轮视觉定位推理的研究者。","confidence":"high","one_line":["VisReason supplies large-scale region-grounded visual CoT traces that teach a global-to-local routine: inspect, localize, zoom, and verify before answering.","VisReason 提供大规模区域定位的视觉 CoT 轨迹，训练模型按“全局观察、定位、放大、核验”的流程作答。"],"why":"Its traces bind intermediate reasoning to image regions, making visual process supervision more inspectable than text-only rationales.","primary_link":"https://arxiv.org/abs/2511.17731","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Y-Research-Group/VisReason"},{"key":"project","label":["Project","项目主页"],"url":"https://lingxiao-li.github.io/visreason.github.io/"}],"link_count":3,"sections":9},{"id":"visual-erm-reward-modeling-visual-equivalence-2026","title":"Visual-ERM: Reward Modeling for Visual Equivalence","year":2026,"venue":"arXiv","authors":["Ziyu Liu","Shengyuan Ding","Xinyu Fang","Xuanlang Dai","Penghui Yang","Jianze Liang","Jiaqi Wang","Kai Chen","Dahua Lin","Yuhang Zang"],"authors_zh":"Ziyu Liu、Shengyuan Ding、Xinyu Fang、Xuanlang Dai、Penghui Yang、Jianze Liang、Jiaqi Wang、Kai Chen、Dahua Lin、Yuhang Zang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","rlvr","test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["vision-to-code","multimodal-reasoning","reward-modeling"],"tags":["vision-to-code","visual-equivalence","reward-model","2026"],"status":"verified","priority":"可读","paper_type_zh":"视觉等价性奖励模型与基准","best_for_zh":"需要视觉到代码 RL 奖励、视觉差异判定或多模态候选修订反馈的研究者","confidence":"high","one_line":["Visual-ERM releases VC-RewardBench and a generative reward model that judges fine-grained visual equivalence between target and rendered code outputs.","VC-RewardBench 含 1,335 个图表、表格与 SVG 视觉等价性实例，以位置、类型、严重度反馈支持视觉代码推理奖励。"],"why":"It evaluates rendered visual equivalence directly, exposing error type and severity rather than relying only on text rules or embedding similarity.","primary_link":"https://arxiv.org/abs/2603.13224","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InternLM/Visual-ERM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/internlm/VC-RewardBench"}],"link_count":4,"sections":9},{"id":"visualprm400k-multimodal-process-supervision-iclr-2026","title":"VISUALPRM400K: An Effective Dataset for Multimodal Process Reward Modeling","year":2026,"venue":"ICLR 2026","authors":["VisualPRM400K authors"],"authors_zh":"VisualPRM400K authors","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"必读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["VISUALPRM400K: An Effective Dataset for Multimodal Process Reward Modeling exposes process or trace supervision data.","发布约 40 万条 VisualPRM400K 多模态过程监督数据，并基于它训练 VisualPRM、构建 VisualProcessBench 评测步骤纠错。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://openreview.net/forum?id=IHyY6vdYZw","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenGVLab/InternVL/tree/main/internvl_chat/shell/internvl3.0/visualprm_data_construction"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenGVLab/VisualPRM400K-v1.1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/OpenGVLab/VisualPRM-8B"},{"key":"project","label":["Project","项目主页"],"url":"https://internvl.github.io/blog/2025-03-13-VisualPRM/"}],"link_count":5,"sections":9},{"id":"vrpo-noisy-reward-value-2026","title":"VRPO: Rethinking Value Modeling for Robust RL under Noisy Supervision in LLM Post-Training","year":2026,"venue":"ACL 2026","authors":["Dingwei Zhu","Shihan Dou","Zhiheng Xi","Senjie Jin","Guoqiang Zhang","Jiazheng Zhang","Junjie Ye","Mingxu Chai","Enyu Zhou","Ming Zhang","Yuhui Wang","Caishuang Huang","Chenhao Huang","Yunke Zhang","Yuran Wang","Tao Gui","Qi Zhang","Xipeng Qiu","Xuanjing Huang"],"authors_zh":"Dingwei Zhu 等（复旦大学、荣耀终端有限公司等）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["rlvr","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["reinforcement_learning","noisy_supervision","preference_learning"],"tags":["rl","noisy-rewards","value-model","advantage-estimation"],"status":"verified","priority":"可读","paper_type_zh":"稳健强化学习目标与价值建模研究","best_for_zh":"适合用带噪规则奖励或学习奖励训练大模型，并需诊断奖励黑客与优势估计不稳定问题的读者。","confidence":"high","one_line":["VRPO makes the value model a noise-regulating part of RL training so uncertain rewards produce more reliable advantage signals.","VRPO 把价值模型变为强化学习中的噪声调节器，使不确定奖励产生更可靠的优势信号。"],"why":"It treats value estimation as the place where noisy training feedback can be filtered before it becomes a policy update.","primary_link":"https://aclanthology.org/2026.acl-long.1103/","links":[],"link_count":2,"sections":9},{"id":"vrprm-process-reward-visual-reasoning-2026","title":"VRPRM: Process Reward Modeling via Visual Reasoning","year":2026,"venue":"ICLR 2026 Poster","authors":["Xinquan Chen","Chongying Yue","Bangwei Liu","Xuhong Wang","Yingchun Wang","Chaochao Lu"],"authors_zh":"Xinquan Chen、Chongying Yue、Bangwei Liu、Xuhong Wang、Yingchun Wang、Chaochao Lu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","process-reward-modeling","visual-reasoning"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要构造多模态 CoT-PRM 监督，或比较小规模解释数据与大规模普通步骤标签协同作用的研究者。","confidence":"high","one_line":["VRPRM uses a small set of teacher-written visual CoT reviews to initialize process reasoning, then scales with ordinary step labels through RL for efficient multimodal reward modeling.","VRPRM 先以少量教师生成的视觉 CoT 评审激活过程推理，再用普通步骤标签强化学习，以较低数据成本训练多模态 PRM。"],"why":"It cleanly separates expensive explanatory supervision from cheaper process labels and releases the former as a reusable training set.","primary_link":"https://arxiv.org/abs/2508.03556","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/two-tiger/VRPRM3.6K"}],"link_count":3,"sections":9},{"id":"webarbiter-web-agent-prm-2026","title":"WebArbiter: A Principle-Guided Reasoning Process Reward Model for Web Agents","year":2026,"venue":"ICLR 2026","authors":["Yao Zhang","Shijie Tang","Zeyu Li","Zhen Han","Volker Tresp"],"authors_zh":"Yao Zhang、Shijie Tang、Zeyu Li、Zhen Han、Volker Tresp","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["process_reward"],"training_use":["agent_training"],"construction_layer":["reward_verifier_layer"],"domains":["agentic-reasoning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"网页智能体过程奖励模型与偏好数据论文","best_for_zh":"需要用可审计的过程反馈训练或搜索网页智能体的读者。","confidence":"high","one_line":["WebArbiter turns web-agent state and competing actions into principle-guided process judgements for preference learning and reward-guided search.","WebArbiter 把网页状态与竞争行动转为由原则引导的过程判断，用于偏好学习和奖励引导的搜索。"],"why":"It makes the connection between a data object and a training objective inspectable.","primary_link":"https://arxiv.org/abs/2601.21872","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YaoZ720/WebArbiterCode"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ZYao720/WebArbiter-Data"},{"key":"project","label":["Project","项目主页"],"url":"https://yaozhang.ai/webarbiter/"}],"link_count":4,"sections":9},{"id":"webchain-2026","title":"WebChain: A Large-Scale Human-Annotated Dataset of Real-World Web Interaction Traces","year":2026,"venue":"CVPR","authors":["Sicheng Fan","Rui Wan","Yifei Leng","Gaoning Liang","Li Ling","Yanyi Shang","Dehan Kong"],"authors_zh":"Sicheng Fan、Rui Wan、Yifei Leng、Gaoning Liang、Li Ling、Yanyi Shang、Dehan Kong","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","frontier_pipeline"],"domains":["web_agents","multimodal_agents"],"tags":["webchain","web-agents","human-trajectories","triple-alignment","gui-grounding","long-horizon-planning","synthetic-rationales","gated-dataset"],"status":"verified","priority":"必读","paper_type_zh":"真实网页人类轨迹数据集与 agent 训练配方","best_for_zh":"关注网页 agent、多模态轨迹、GUI grounding、长程规划，以及开放发布中许可、隐私与可重放性审计的研究者","confidence":"medium","one_line":["Releases 31,725 human-verified live-web trajectories with 317,993 visually, structurally, and action-aligned steps, plus separate grounding and planning recipes, under a gated academic license.","WebChain 发布 31,725 条人工核验的真实网页轨迹，包含 317,993 个视觉、结构与动作对齐的步骤，并以 gated 学术许可提供 grounding 与长程规划训练材料。"],"why":"For the open-release-recipe track, WebChain makes the training unit and three-stage construction pipeline concrete, but also shows why artifact completeness, success semantics, source rights, privacy evidence, and replay metadata must be audited independently of benchmark gains.","primary_link":"https://openaccess.thecvf.com/content/CVPR2026/papers/Fan_WebChain_A_Large-Scale_Human-Annotated_Dataset_of_Real-World_Web_Interaction_Traces_CVPR_2026_paper.pdf","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/webagentlab/WebChain"}],"link_count":4,"sections":9},{"id":"webdevjudge-2026","title":"WebDevJudge: Evaluating (M)LLMs as Critiques for Web Development Quality","year":2026,"venue":"ICLR 2026 Oral","authors":["Chunyang Li","Yilun Zheng","Xinting Huang","Tianqing Fang","Jiahao Xu","Lihui Chen","Yangqiu Song","Han Hu"],"authors_zh":"Chunyang Li、Yilun Zheng、Xinting Huang、Tianqing Fang、Jiahao Xu、Lihui Chen、Yangqiu Song、Han Hu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["reasoning","evaluation","agents"],"tags":["track07","judge","web-development"],"status":"verified","priority":"可读","paper_type_zh":"网页开发 LLM-as-a-judge 元评测与可行性诊断基准","best_for_zh":"需要审计网页代码 critic、agent judge 或网页 reward signal 的研究者。","confidence":"high","one_line":["An ICLR 2026 oral benchmark for evaluating model critiques of web-development quality.","以部署网页、rubric tree 与专家偏好审计网页开发 judge 的基准。"],"why":"It provides a domain-specific judgment surface for studying evaluator reliability.","primary_link":"https://openreview.net/forum?id=CCSPm6V5EF","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lcy2723/WebDevJudge"}],"link_count":2,"sections":9},{"id":"webforge-browser-agent-benchmark-2026","title":"WebForge: Breaking the Realism-Reproducibility-Scalability Trilemma in Browser Agent Benchmark","year":2026,"venue":"arXiv preprint","authors":["Peng Yuan","Yuyang Yin","Yuxuan Cai","Zheng Wei"],"authors_zh":"Peng Yuan 等（Tencent BAC、Tsinghua University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","synthetic-web-environments"],"tags":["agent_environment","benchmark","evaluation-surface","synthetic-web-environments","trajectory_data","web-browser-agents"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的评测面 / 环境基准","best_for_zh":"关注浏览器智能体、合成网页环境、可复现实验和环境轨迹审计的读者。","confidence":"high","one_line":["WebForge generates self-contained browser environments and WebForge-Bench tasks to evaluate browser agents without relying on drifting live websites.","WebForge 生成自包含浏览器环境和 WebForge-Bench 任务，用来在不依赖易漂移真实网站的情况下评测浏览器智能体。"],"why":"WebForge generates self-contained browser environments and WebForge-Bench tasks to evaluate browser agents without relying on drifting live websites.","primary_link":"https://arxiv.org/abs/2604.10988","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yuandaxia2001/WebForge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yuandaxia/WebForge"}],"link_count":4,"sections":9},{"id":"webseer-2025","title":"WebSeer: Training Deeper Search Agents through Reinforcement Learning with Self-Reflection","year":2026,"venue":"ICLR 2026","authors":["Guanzhong He","Zhen Yang","Jinxin Liu","Bin Xu","Lei Hou","Juanzi Li"],"authors_zh":"Guanzhong He, Zhen Yang, Jinxin Liu, Bin Xu, Lei Hou, Juanzi Li","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","agent_training","test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","raw_search_rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["WebSeer constructs verified self-reflective SFT trajectories and trains a web-search agent with multi-submission RL feedback, while releasing code, two datasets, and a 14B checkpoint with incomplete evaluation tooling.","WebSeer 结合反思式拒绝采样与在线强化学习训练深度搜索代理，并链接数据、代码和模型。"],"why":"It exposes how failed answer submissions can become continued search-and-reflection interactions, making the trajectory schema, feedback contract, environment drift, and replay risks concrete for rollout and test-time trace curation.","primary_link":"https://openreview.net/pdf?id=YCXWIfVakj","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/99hgz/WebSeer"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/99hgz/WebSeer-dataset"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/99hgz/WebSeer-14b"}],"link_count":7,"sections":9},{"id":"webstar-scalable-computer-use-step-filtering-2026","title":"WebSTAR: Scalable Data Synthesis for Computer Use Agents with Step-Level Filtering","year":2026,"venue":"ACL 2026","authors":["Yifei He","Pranit Chawla","Yaser Souri","Subhojit Som","Xia Song"],"authors_zh":"Yifei He, Pranit Chawla, Yaser Souri, Subhojit Som, Xia Song","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["computer-use","web-agents","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"计算机使用智能体的合成轨迹与逐步质量过滤数据集论文","best_for_zh":"需要训练、筛选或评测网页操作智能体轨迹及逐步反馈信号的研究者。","confidence":"high","one_line":["WebSTAR releases scalable synthetic computer-use trajectories together with WebSCORE step-level quality feedback for training web agents.","以网页交互轨迹合成与逐步过滤构建 WebSTAR 和 WebSCORE，为计算机使用智能体提供带细粒度质量信号的大规模训练数据。"],"why":"It makes trajectory generation and step-level filtering available as a single public computer-use data resource.","primary_link":"https://aclanthology.org/2026.acl-long.21/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/microsoft/WebSTAR"},{"key":"project","label":["Project","项目主页"],"url":"https://yifei-he.github.io/webstar-website/"}],"link_count":5,"sections":9},{"id":"webworld-2026","title":"WebWorld: A Large-Scale World Model for Web Agent Training","year":2026,"venue":"arXiv preprint","authors":["Zikai Xiao","Jianhong Tu","Chuhang Zou","Yuxin Zuo","Zhi Li","Peng Wang","Bowen Yu","Fei Huang","Junyang Lin","Zuozhu Liu"],"authors_zh":"Zikai Xiao 等（机构以 arXiv 正文为准）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","world-models"],"tags":["benchmark","agent_environment","web-browser-agents","world-models"],"status":"verified","priority":"可读","paper_type_zh":"arXiv 的 web world-model paper","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"medium","one_line":["WebWorld trains large web world models for web-agent simulation, synthetic trajectory generation, and search.","WebWorld 训练网页世界模型，用于网页智能体仿真、合成轨迹生成和搜索。"],"why":"it explores world-model simulation as a route to long-horizon web environments","primary_link":"https://arxiv.org/abs/2602.14721","links":[],"link_count":2,"sections":9},{"id":"wici-instruction-data-2026","title":"What Makes Good Instruction-Tuning Data? An In-Context Learning Perspective","year":2026,"venue":"ACL 2026","authors":["Guangzeng Han","Xiaolei Huang"],"authors_zh":"Guangzeng Han, Xiaolei Huang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"可读","paper_type_zh":"训练数据选择或奖励设计论文","best_for_zh":"研究后训练数据消费与优化目标的读者。","confidence":"high","one_line":["wICI ranks instruction data by how much a candidate demonstration reduces difficulty for semantically related peers.","wICI 按候选示例降低语义相近样本难度的程度为指令数据排序。"],"why":"It makes the training-data decision inspectable.","primary_link":"https://aclanthology.org/2026.acl-long.45/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/trust-nlp/SyntheticData-Curator"}],"link_count":2,"sections":9},{"id":"when-benchmarks-leak-2026","title":"When Benchmarks Leak: Inference-Time Decontamination for LLMs","year":2026,"venue":"ACL 2026","authors":["Chai Jianzhe","Yu Zhe","Jun Sakuma"],"authors_zh":"Chai Jianzhe, Yu Zhe, Jun Sakuma","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["contamination","verifier-audit","evaluation-reliability"],"tags":["track13","audit","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"推理时基准去污染与评测可靠性研究","best_for_zh":"需要在保留原始基准的前提下审计泄露分数与干净效用的研究者。","confidence":"medium","one_line":["DeconIEP suppresses memorization-driven behavior at evaluation time with bounded embedding perturbations.","以参考模型引导的有界嵌入扰动，在不改题目或参数的前提下抑制污染捷径。"],"why":"It separates a contaminated model’s shortcut behavior from clean-input utility without deleting benchmark items.","primary_link":"https://aclanthology.org/2026.acl-long.2071/","links":[],"link_count":1,"sections":9},{"id":"overthinking-test-time-compute-2026","title":"When More Thinking Hurts: Overthinking in LLM Test-Time Compute Scaling","year":2026,"venue":"Findings of ACL 2026","authors":["Shu Zhou","Rui Ling","Junan Chen","Xin Wang","Tao Fan","Hao Wang"],"authors_zh":"Shu Zhou、Rui Ling、Junan Chen、Hao Wang（南京大学）；Xin Wang（百度）；Tao Fan（南京财经大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report","optimizer_scaffold"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-scaling","overthinking","adaptive-stopping","compute-budget","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"测试时计算扩展评测与效率分析","best_for_zh":"需要选择停止预算，或评估自适应推理系统且希望得到准确率—成本曲线而非单一最大预算分数的读者。","confidence":"high","one_line":["This study shows that longer reasoning can reverse correct answers and proposes marginal-utility and flip-event analysis for selecting test-time budgets.","这项研究显示，更长的推理会把原本正确的答案改错，并用边际效用和答案翻转分析来选择测试时预算。"],"why":"It supplies a concrete failure mode for indiscriminate test-time scaling and turns “when to stop” into a measurable allocation decision.","primary_link":"https://aclanthology.org/2026.findings-acl.1199/","links":[],"link_count":2,"sections":9},{"id":"drift-telbench-agent-trajectory-errors-2026","title":"Where Do Deep-Research Agents Go Wrong? Span-Level Error Localization in Agent Trajectories","year":2026,"venue":"arXiv preprint","authors":["Jiaming Wang","Ziteng Feng","Jiangtao Wu","Ruihao Li","Qianqian Xie","Yuxiang Ren","He Zhu","Xueming Han","Fanyu Meng","Junlan Feng","Jiaheng Liu"],"authors_zh":"Jiaming Wang、Ziteng Feng、Jiangtao Wu、Ruihao Li、Qianqian Xie、Yuxiang Ren、He Zhu、Xueming Han、Fanyu Meng、Junlan Feng、Jiaheng Liu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["deep-research-agents","trajectory-auditing","process-evaluation"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要诊断研究代理轨迹、训练过程审计器或评估首错定位能力的研究者。","confidence":"high","one_line":["TELBench labels harmful spans in deep-research trajectories, while DRIFT traces unsupported claims to localize where an agent first goes wrong.","TELBench 标注深度研究代理轨迹中的有害错误片段，DRIFT 通过追踪无证据主张定位代理最早出错的位置。"],"why":"It turns opaque final-answer failures into localized, evidence-linked process errors that can supervise trajectory auditing.","primary_link":"https://arxiv.org/abs/2606.02060","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/NJU-LINK/TELBench"},{"key":"project","label":["Project","项目主页"],"url":"https://nju-link.github.io/DRIFT/"}],"link_count":4,"sections":9},{"id":"rsr-informative-alignment-2026","title":"Which Reasoning Trajectories Teach Students to Reason Better? A Simple Metric of Informative Alignment","year":2026,"venue":"ACL 2026","authors":["Yuming Yang","Mingyoung Lai","Wanxu Zhao","Xiaoran Fan","Zhiheng Xi","Mingqi Wu","Chiyue Huang","Jun Zhao","Haijun Lv","Jian Tong","Yunhua Zhou","Yicheng Zou","Qipeng Guo","Tao Gui","Qi Zhang","Xuanjing Huang"],"authors_zh":"Yuming Yang、Mingyoung Lai、Wanxu Zhao、Xiaoran Fan、Zhiheng Xi、Mingqi Wu、Chiyue Huang、Jun Zhao、Haijun Lv、Jian Tong、Yunhua Zhou、Yicheng Zou、Qipeng Guo、Tao Gui、Qi Zhang、Xuanjing Huang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","scaling_study","audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","trajectory_value"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","scaling_report","release_audit"],"domains":["math","reasoning-distillation"],"tags":["rank-surprisal-ratio","informative-alignment","trajectory-selection","teacher-selection","student-conditioned-data","reasoning-distillation","long-cot","open-release"],"status":"partial","priority":"可读","paper_type_zh":"学生条件化轨迹选择与开放数据发布","best_for_zh":"研究推理蒸馏、数据选择、多教师轨迹和决策谱系审计的读者","confidence":"medium","one_line":["RSR scores each teacher trajectory under the target student and selects one of 33 candidates per math problem by balancing clipped token rank against surprisal.","RSR 在目标学生模型下权衡 token rank 与 surprisal，从每题 33 条教师轨迹中选择一条；它衡量适配性而非正确性。"],"why":"It operationalizes student-specific reasoning-data construction and releases both the candidate pool and five selected datasets, while showing why selection quality, correctness, contamination and decision-lineage records must be audited separately.","primary_link":"https://aclanthology.org/2026.acl-long.1950/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UmeanNever/RankSurprisalRatio"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Umean/RSR_data"}],"link_count":6,"sections":9},{"id":"french-medical-judge-2026","title":"Who Judges the Judge? Evaluating LLM-as-a-Judge for French Medical open-ended QA","year":2026,"venue":"HeaLing 2026","authors":["Ikram Belmadani","Oumaima El Khettari","Pacôme Constant dit Beaufils","Richard Dufour","Benoit Favre"],"authors_zh":"Ikram Belmadani, Oumaima El Khettari, Pacôme Constant dit Beaufils, Richard Dufour, Benoit Favre","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"法语医疗开放式问答中的 LLM 评判可靠性研究","best_for_zh":"需要以专家标注审计医疗 LLM 评判器的人。","confidence":"medium","one_line":["Evaluates generator-aware, domain-specific judge reliability for medical open-ended QA.","在法语开放式医疗问答中，LLM 评判会随答案生成器而偏移；小模型经 SFT 与 GRPO 对齐可改善但不能替代专家。"],"why":"It makes generator-dependent failure modes visible in domain-specific LLM evaluation.","primary_link":"https://aclanthology.org/2026.healing-1.12/","links":[],"link_count":1,"sections":9},{"id":"who-when-pro-2026","title":"Who&When Pro: Can LLMs Really Attribute Failures in AI Agents?","year":2026,"venue":"arXiv preprint","authors":["Jiale Liu","Huajun Xi","Shaokun Zhang","Yifan Zeng","Tianwei Yue","Chi Wang","Jian Kang","Qingyun Wu","Huazheng Wang"],"authors_zh":"Jiale Liu、Huajun Xi、Shaokun Zhang、Yifan Zeng、Tianwei Yue、Chi Wang、Jian Kang、Qingyun Wu、Huazheng Wang","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","construction_recipe","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["step_level","state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","failure_attribution","multi_agent_systems","multimodal_agents","web_and_tool_use","code_and_data_science","gui_and_embodied_interaction","video_understanding"],"tags":["who-when-pro","agent-failure-attribution","controlled-error-injection","decisive-step","multi-agent-systems","multimodal-trajectories","mixed-evaluator","warm-start-replay","benchmark-audit","unreleased-data","paper-release-discrepancy"],"status":"partial","priority":"可读","paper_type_zh":"多模态智能体失败归因基准与构造流程","best_for_zh":"研究智能体轨迹、失败归因、因果注入、评测器与多模态长上下文审计的读者","confidence":"medium","one_line":["Who&When Pro reports 12,326 controlled failed trajectories across 3 modalities and 26 benchmarks with responsible-agent, decisive-step, and error-mode labels, while the official data and evaluation code remain unavailable.","Who&When Pro 报告 12,326 条跨 3 种模态与 26 个基准的受控失败轨迹，并提供责任智能体、决定性步骤和错误模式标签，但官方数据与评测代码尚未公开。"],"why":"It turns agent debugging into a concrete episode-level evaluation contract and tests whether frontier models can localize causal failures, while demonstrating why successful anchors, discarded injections, evaluator versions, replay state, label corrections, and actual release artifacts must be audited before a benchmark can support training or reproducible evaluation.","primary_link":"https://arxiv.org/abs/2607.09996","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/whowhenpro/whowhen_pro"},{"key":"project","label":["Project","项目主页"],"url":"https://whowhenpro.github.io/"}],"link_count":6,"sections":9},{"id":"wildreward-human-interactions-2026","title":"WildReward: Learning Reward Models from In-the-Wild Human Interactions","year":2026,"venue":"arXiv","authors":["Hao Peng","Junjia Qi","Xiaozhi Wang","Yijun Yao","Lei Hou","Yuanzi Li"],"authors_zh":"Hao Peng、Junjia Qi、Xiaozhi Wang、Yijun Yao、Lei Hou、Yuanzi Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"真实人类交互奖励数据与奖励模型论文","best_for_zh":"研究偏好数据、奖励建模或对齐的读者。","confidence":"medium","one_line":["The paper provides preference or reward feedback data for alignment research.","WildReward 从真实人类交互中学习奖励模型，提供面向开放场景的偏好反馈数据。"],"why":"It makes a feedback object available for alignment training, evaluation, or analysis.","primary_link":"https://arxiv.org/abs/2602.08829","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/WildReward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THU-KEG/WildFB"}],"link_count":3,"sections":9},{"id":"workforceagent-r1-2025","title":"WorkForceAgent-R1: Incentivizing Reasoning Capability in LLM-based Web Agents via Reinforcement Learning","year":2026,"venue":"Findings of the Association for Computational Linguistics: EACL 2026","authors":["Yuchen Zhuang","Di Jin","Jiaao Chen","Wenqi Shi","Hanrui Wang","Chao Zhang"],"authors_zh":"Yuchen Zhuang、Di Jin、Jiaao Chen、Wenqi Shi、Hanrui Wang、Chao Zhang","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","agent_environment","verifier_reward","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","environment_interaction","web_navigation","workplace_automation"],"tags":["workforceagent-r1","web-agent","workarena","browsergym","oracle-trajectories","single-step-reasoning","grpo","rule-based-reward","release-gap"],"status":"partial","priority":"可读","paper_type_zh":"网页智能体强化学习与规则验证方法","best_for_zh":"关注 environment-agent trajectory、WorkArena、RLVR verifier、失败样本保留与复现审计的读者","confidence":"high","one_line":["WorkForceAgent-R1 turns filtered WorkArena oracle trajectories into single-step action-verification problems for SFT and GRPO, but releases code without the trajectory corpus, rejection ledger, or trained checkpoints.","WorkForceAgent-R1 将筛选后的 WorkArena oracle 轨迹拆成单步动作验证样本，用规则奖励进行 SFT 与 GRPO，但未公开轨迹语料、淘汰记录或训练后模型。"],"why":"It isolates a cheap rule-based feedback contract for workplace web-agent reasoning while showing why paper-scale configurations, derived records, public artifacts, proxy rewards, and complete success/failure retention must be audited separately.","primary_link":"https://aclanthology.org/2026.findings-eacl.3/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/night-chen/WorkForceAgent-R1"},{"key":"project","label":["Project","项目主页"],"url":"https://www.hanruiwang.com/projects/workforceagent-r1"}],"link_count":8,"sections":9},{"id":"writing-rl-adaptive-curriculum-2026","title":"Writing-RL: Advancing Long-form Writing via Adaptive Curriculum Reinforcement Learning","year":2026,"venue":"ACL 2026 Long Papers","authors":["Xuanyu Lei","Chenliang Li","Yuning Wu","Kaiming Liu","Weizhou Shen","Peng Li","Ming Yan","Fei Huang","Ya-Qin Zhang","Yang Liu"],"authors_zh":"Xuanyu Lei, Chenliang Li, Yuning Wu, Kaiming Liu, Weizhou Shen, Peng Li, Ming Yan, Fei Huang, Ya-Qin Zhang, Yang Liu","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["long-form-writing","instruction-tuning"],"tags":["long-form-writing","reinforcement-learning","curriculum","data-selection","pairwise-reward"],"status":"verified","priority":"必读","paper_type_zh":"面向长篇写作的自适应课程强化学习研究","best_for_zh":"为开放式生成设计数据选择和自适应参考调度的读者。","confidence":"high","one_line":["Writing-RL selects long-form writing instructions by policy-to-reference quality margin and promotes the reference after each policy win.","该工作按策略与参考写作质量差选择长文指令，并在策略获胜后逐步提升参考答案难度。"],"why":"It makes the training record and reward target progressively harder instead of holding a static preference target throughout RL.","primary_link":"https://aclanthology.org/2026.acl-long.255/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tongyi-Zhiwen/Writing-RL"}],"link_count":3,"sections":9},{"id":"you-only-judge-once-multi-response-rm-2026","title":"You Only Judge Once: Multi-response Reward Modeling in a Single Forward Pass","year":2026,"venue":"arXiv 2026","authors":["Yinuo Yang","Zixian Ma","Manasi Ganti","Jieyu Zhang","Ranjay Krishna"],"authors_zh":"Yinuo Yang, Zixian Ma, Manasi Ganti, Jieyu Zhang, Ranjay Krishna","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["vision-language","reward-modeling","preference-ranking"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多响应多模态奖励模型与人类偏好排序基准论文","best_for_zh":"希望以一次前向比较多个图像或视频候选回答、并需要公开 N 路人类排序数据的奖励模型研究者。","confidence":"high","one_line":["MR²Bench turns multimodal response evaluation into an N-way human-ranking task and supports a reward model that scores all candidates jointly in one forward pass.","MR²Bench 含图像多响应排序与视频问答约 9.4 万众包偏好判断，可直接训练一次前向完成 N 路比较的奖励模型。"],"why":"It exposes ranking information that pairwise-only reward benchmarks discard.","primary_link":"https://arxiv.org/abs/2604.10966","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yinuoyang01/multi-response-rm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yinuoy/MR2Bench"}],"link_count":4,"sections":9},{"id":"zebraarena-a-diagnostic-simulation-environment-for-studying-reasoning-action-coupling-in","title":"ZEBRAARENA: A Diagnostic Simulation Environment for Studying Reasoning-Action Coupling in Tool-Augmented LLMs","year":2026,"venue":"arXiv","authors":["Wanjia Zhao","Ludwig Schmidt","James Zou","Vidhisha Balachandran","Lingjiao Chen"],"authors_zh":"Wanjia Zhao, Ludwig Schmidt, James Zou, Vidhisha Balachandran, Lingjiao Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A procedurally generated Zebra tool environment with unique solutions and optimal query counts for reasoning–action coupling.","程序生成的 Zebra 工具环境，用唯一解和最优查询数测推理—行动耦合。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2603.18614","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wanjiaZhao1203/ZebraArena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ZebraArena/ZebraArena"}],"link_count":3,"sections":9},{"id":"zero-to-cad-2026","title":"Zero-to-CAD: Agentic Synthesis of Interpretable CAD Programs at Million-Scale Without Real Data","year":2026,"venue":"arXiv preprint","authors":["Mohammadmehdi Ataei","Farzaneh Askari","Kamal Rahimi Malekshan","Pradeep Kumar Jayaraman"],"authors_zh":"Mohammadmehdi Ataei、Farzaneh Askari、Kamal Rahimi Malekshan、Pradeep Kumar Jayaraman","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["executable-CAD-construction-demonstrations"],"tags":["instruction-demonstration-rationale","arxiv-2604.24479","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"CAD 程序生成监督微调与代理训练","confidence":"high","one_line":["Zero-to-CAD uses tool-using agents to synthesize, execute, repair, and validate nearly one million CadQuery construction programs without real trajectories.","Zero-to-CAD 用工具代理生成、执行、修复并验证近百万条可编辑的 CAD 构造程序。"],"why":"Real CAD datasets expose final geometry but rarely the editable construction history needed to train interpretable program-generating models.","primary_link":"https://arxiv.org/abs/2604.24479","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ADSKAILab/Zero-To-CAD-1m"}],"link_count":2,"sections":9},{"id":"zipping-the-thought-2026","title":"Zipping the Thought: When and How Compressed Reasoning Data Works in LLM Post-Training","year":2026,"venue":"arXiv preprint; under review (2026)","authors":["Kohsei Matsutani","Gouki Minegishi","Takeshi Kojima","Yusuke Iwasawa","Yutaka Matsuo"],"authors_zh":"Kohsei Matsutani, Gouki Minegishi, Takeshi Kojima, Yusuke Iwasawa, Yutaka Matsuo","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["sft","rlvr"],"construction_layer":["trace_writing","frontier_pipeline"],"domains":["compositional-reasoning"],"tags":["cot-compression","controlled-synthetic-data","sft-rlvr"],"status":"verified","priority":"可读","paper_type_zh":"压缩推理数据的受控研究","best_for_zh":"适合选择 CoT 粒度及 SFT、RLVR 数据预算的研究者。","confidence":"high","one_line":["Zipping the Thought varies CoT compression in controlled synthetic data to show how data diversity, repetition, and RLVR affect step decomposition and length generalization.","Zipping the Thought 在受控合成数据中改变 CoT 压缩粒度，分析数据多样性、重复训练与 RLVR 如何影响步骤拆解和长度泛化。"],"why":"It separates trace granularity from data volume and optimization, giving concrete rules for designing compressed reasoning supervision.","primary_link":"https://arxiv.org/abs/2605.28008","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kohseim/cot_compression"}],"link_count":2,"sections":9},{"id":"nabla-reasoner-2026","title":"∇-Reasoner: LLM Reasoning via Test-Time Gradient Descent in Textual Space","year":2026,"venue":"ICLR 2026","authors":["Peihao Wang","Ruisi Cai","Zhen Wang","Hongyuan Mei","Qiang Liu","Pan Li","Zhangyang Wang"],"authors_zh":"Peihao Wang、Ruisi Cai、Zhen Wang、Hongyuan Mei、Qiang Liu、Pan Li、Zhangyang Wang（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（ICLR 2026）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["∇-Reasoner: LLM Reasoning via Test-Time Gradient Descent in Textual Space","离散搜索和试错提示可能花费许多测试时调用，却只能间接改善推理策略。"],"why":"It makes an inference-budget decision auditable.","primary_link":"https://iclr.cc/virtual/2026/poster/10007349","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/VITA-Group/Nabla-Reasoner"}],"link_count":2,"sections":9},{"id":"tau2-bench-2025","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","year":2025,"venue":"ICML 2026","authors":["Victor Barres","Honghua Dong","Soham Ray","Xujie Si","Karthik Narasimhan"],"authors_zh":"Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["conversational_agents","customer_service","tool_use","environment_interaction","telecom","airline","retail"],"tags":["environment-agent-trajectory-data","agent-benchmark","dual-control","tool-use","user-simulator","compositional-task-generation","programmatic-verification","trajectory-release","version-drift"],"status":"partial","priority":"必读","paper_type_zh":"双控制客服智能体评测环境与轨迹发布","best_for_zh":"研究环境智能体轨迹、用户模拟器、程序化 verifier 与 benchmark 版本审计的读者","confidence":"high","one_line":["tau2-bench releases a dual-control telecom environment with 114 sampled tasks from 2,285 generated compositions, state/action-grounded rewards, and full success/failure evaluation trajectories, but exact reuse requires a paper-era version pin.","tau2-bench 发布双控制客服环境，从 2,285 个 telecom 完整组合中抽取 114 个评测任务，并保留状态/动作反馈及成功与失败轨迹；准确复用必须固定论文时代的 v0.1.0。"],"why":"It turns agent-user coordination into an auditable data object—task state, messages, both parties’ tool actions, observations, verifier components, reward, and termination—while showing why public environments still need version, rights, simulator, and contamination review before trajectories become post-training data.","primary_link":"https://openreview.net/pdf?id=OC2z7iSQKa","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sierra-research/tau2-bench"},{"key":"data","label":["Data","数据"],"url":"https://github.com/sierra-research/tau2-bench/tree/v0.1.0/data/tau2"},{"key":"project","label":["Project","项目主页"],"url":"https://sierra.ai/resources/research/tau-squared-bench"}],"link_count":10,"sections":9},{"id":"am-deepseek-r1-distilled-2025","title":"1.4 Million Open-Source Distilled Reasoning Dataset to Empower Large Language Model Training","year":2025,"venue":"arXiv preprint (2025)","authors":["Han Zhao","Haotian Wang","Yiping Peng","Sitong Zhao","Xiaoyu Tian","Shuaiting Chen","Yunjie Ji","Xiangang Li"],"authors_zh":"Han Zhao、Haotian Wang、Yiping Peng、Sitong Zhao、Xiaoyu Tian、Shuaiting Chen、Yunjie Ji、Xiangang Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["Chinese and English mathematics, code, science, and general reasoning"],"tags":["instruction-demonstration-rationale","arxiv-2503.19633","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"推理监督微调和知识蒸馏","confidence":"high","one_line":["AM-Distilled unifies 1.4 million R1 responses and routes math, code, and other tasks through answer checks, execution, or reward-model review.","AM-Distilled 把 140 万条中英双语 R1 推理按数学答案、代码测试或奖励模型分别验证。"],"why":"Large reasoning teachers are useful for distillation, but public students lack a bilingual, cross-domain corpus whose verification rule changes with the task type.","primary_link":"https://arxiv.org/abs/2503.19633","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/a-m-team/AM-DeepSeek-R1-Distilled-1.4M"}],"link_count":2,"sections":9},{"id":"3ds-medical-domain-selection-2025","title":"3DS: Medical Domain Adaptation of LLMs via Decomposed Difficulty-based Data Selection","year":2025,"venue":"EMNLP 2025","authors":["Hongxin Ding","Yue Fang","Runchuan Zhu","Xinke Jiang","Jinyang Zhang","Yongxin Xu","Weibin Liao","Xu Chu","Junfeng Zhao","Yasha Wang"],"authors_zh":"Hongxin Ding, Yue Fang, Runchuan Zhu, Xinke Jiang, Jinyang Zhang, Yongxin Xu, Weibin Liao, Xu Chu, Junfeng Zhao, Yasha Wang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["medical","instruction-tuning","law","general"],"tags":["medical","data-selection","difficulty","sft"],"status":"verified","priority":"必读","paper_type_zh":"以模型为中心的领域指令数据选择研究","best_for_zh":"把已知模型适配到带噪专业指令池的读者。","confidence":"high","one_line":["3DS filters domain SFT records by target-model quality, decomposed moderate difficulty, and diversity.","3DS 按目标模型质量、分解后的适度难度和多样性过滤领域监督微调记录。"],"why":"It makes target-model knowledge compatibility and learnable difficulty explicit before records are admitted to SFT.","primary_link":"https://aclanthology.org/2025.emnlp-main.983/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PuppyKnightUniversity/3DS"}],"link_count":3,"sections":9},{"id":"vericoding-formally-verified-program-synthesis-2025","title":"A benchmark for vericoding: formally verified program synthesis","year":2025,"venue":"Dafny 2026 at POPL 2026","authors":["Sergiu Bursuc","Theodore Ehrenborg","Shaowei Lin","Lacramioara Astefanoaei","Ionel Emilian Chiosa","Jure Kukovec","Alok Singh","Oliver Butterley","Adem Bizid","Quinn Dougherty","Miranda Zhao","Max Tan","Max Tegmark"],"authors_zh":"Sergiu Bursuc, Theodore Ehrenborg, Shaowei Lin, Lacramioara Astefanoaei, Ionel Emilian Chiosa, Jure Kukovec, Alok Singh, Oliver Butterley, Adem Bizid, Quinn Dougherty, Miranda Zhao, Max Tan, Max Tegmark","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","code-generation","lean4"],"tags":["formal-verification","dafny","verus","lean4","2025"],"status":"verified","priority":"可读","paper_type_zh":"跨语言形式验证程序综合基准","best_for_zh":"需要比较 Dafny、Verus/Rust 与 Lean 上可验证代码生成能力的研究者。","confidence":"high","one_line":["Vericoding aggregates 12,504 Dafny, Verus, and Lean specifications whose generated code and proofs can be replayed against formal verifiers.","Vericoding 汇集 12,504 个 Dafny、Verus 与 Lean 规格任务，并可用相应形式验证器复跑生成的代码与证明。"],"why":"It makes successful program synthesis a proof-checking outcome across multiple formal ecosystems.","primary_link":"https://arxiv.org/abs/2509.22908","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Beneficial-AI-Foundation/vericoding-benchmark"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/beneficial-ai-foundation/vericoding"}],"link_count":5,"sections":9},{"id":"survey-reward-models-2025","title":"A Comprehensive Survey of Reward Models: Taxonomy, Applications, Challenges, and Future","year":2025,"venue":"arXiv preprint","authors":["Jialun Zhong","Wei Shen","Yanzeng Li","Songyang Gao","Hua Lu","Yicheng Chen","Yang Zhang","Wei Zhou","Jinjie Gu","Lei Zou"],"authors_zh":"Jialun Zhong 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["scalar_reward","pairwise_preference"],"training_use":["reward_modeling","preference_learning","safety_alignment"],"construction_layer":["release_audit"],"domains":["reward-modeling","alignment","reasoning-data"],"tags":["reward-model-survey","preference-data","RLHF","alignment"],"status":"verified","priority":"必读","paper_type_zh":"奖励模型与偏好数据综述","best_for_zh":"需要厘清偏好对、标量奖励、奖励模型与裁判模型职责边界的读者。","confidence":"high","one_line":["A 2025 reward-model map that separates feedback collection, reward fitting, and downstream use.","梳理反馈收集、奖励建模、奖励使用与评测边界的 2025 年奖励模型综述。"],"why":"Reward language is often overloaded; this survey routes readers to the actual data or verifier object they need to audit.","primary_link":"https://arxiv.org/abs/2504.12328","links":[],"link_count":2,"sections":9},{"id":"learning-from-rewards-llm-2025","title":"A Comprehensive Survey on Learning from Rewards for Large Language Models: Reward Models and Learning Strategies","year":2025,"venue":"Findings of EMNLP 2025","authors":["Xiaobao Wu"],"authors_zh":"Xiaobao Wu","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["preference_learning","reward_modeling","test_time_compute"],"construction_layer":["reward_verifier_layer","optimizer_scaffold"],"domains":["reward-modeling","preference-learning","post-training","reasoning-data"],"tags":["foundations-and-primers","reward-models","preference-data","learning-from-rewards","emnlp-2025"],"status":"verified","priority":"可读","paper_type_zh":"奖励模型与学习策略综述","best_for_zh":"关于偏好对、标量奖励、奖励模型和学习策略的综述。","confidence":"high","one_line":["A 2025 EMNLP Findings survey that maps reward signals from feedback collection through training, inference, and post-inference correction.","一篇把奖励信号从反馈收集延伸到训练、推理与后处理的 2025 EMNLP Findings 综述。"],"why":"It prevents Track0 readers from collapsing preference data, reward models, optimization objectives, and judge outputs into one undifferentiated “RLHF” label.","primary_link":"https://aclanthology.org/2025.findings-emnlp.970/","links":[],"link_count":3,"sections":9},{"id":"imo-steps-lean-dataset-2025","title":"A Lean Dataset for International Math Olympiad: Small Steps towards Writing Math Proofs for Hard Problems","year":2025,"venue":"Transactions on Machine Learning Research","authors":["Roozbeh Yousefzadeh","Xuenan Cao","Azim Ospanov"],"authors_zh":"Roozbeh Yousefzadeh, Xuenan Cao, Azim Ospanov","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","olympiad-math"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"必读","paper_type_zh":"国际奥赛 Lean 分步证明数据集","best_for_zh":"需要以中间引理分析 Lean 证明能力或训练形式化证明模型的研究者。","confidence":"high","one_line":["IMO-Steps decomposes International Mathematical Olympiad Lean proofs into 1,329 checkable intermediate lemmas and complete proofs.","IMO-Steps 将国际奥赛题的 Lean 证明拆分为可单独验证的引理与完整证明，便于诊断证明器的局部失败。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://openreview.net/forum?id=CrKMqRAhBo","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/roozbeh-yz/IMO-Steps"}],"link_count":2,"sections":9},{"id":"a-sober-look-at-progress-in-language-model-reasoning-pitfalls-and-paths-to-repro-2025","title":"A Sober Look at Progress in Language Model Reasoning: Pitfalls and Paths to Reproducibility","year":2025,"venue":"Conference on Language Modeling (COLM)","authors":["Andreas Hochlehnert","Hardik Bhatnagar","Vishaal Udandarao","Samuel Albanie","Ameya Prabhu","Matthias Bethge"],"authors_zh":"Andreas Hochlehnert, Hardik Bhatnagar, Vishaal Udandarao, Samuel Albanie, Ameya Prabhu, Matthias Bethge","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"必读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Audits reasoning-model progress claims by showing that benchmark results can be highly sensitive to decoding, seeds, prompt format, and environment details.","指出解码、随机种子、提示格式和运行环境都可能改变推理结论，是复现与比较模型进展的基础审计读物。"],"why":"It is an audit anchor for this atlas: reasoning-data claims need reproducible evaluation settings, not just headline benchmark gains.","primary_link":"https://arxiv.org/abs/2504.07086","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bethgelab/sober-reasoning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/bethgelab/sober_reasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://bethgelab.github.io/sober-reasoning/"}],"link_count":4,"sections":9},{"id":"llm-reasoning-frontiers-survey-2025","title":"A Survey of Frontiers in LLM Reasoning: Inference Scaling, Learning to Reason, and Agentic Systems","year":2025,"venue":"Transactions on Machine Learning Research","authors":["Zixuan Ke","Fangkai Jiao","Yifei Ming","Xuan-Phi Nguyen","Austin Xu","Do Xuan Long","Minzhi Li","Chengwei Qin","Peifeng Wang","Silvio Savarese","Caiming Xiong","Shafiq Joty"],"authors_zh":"Zixuan Ke 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["trace_writing"],"domains":["reasoning","reasoning-data","inference-scaling","reinforcement-learning","agents"],"tags":["foundations-and-primers","llm-reasoning","agentic-systems","tmlr-2025","survey"],"status":"verified","priority":"必读","paper_type_zh":"大模型推理综述","best_for_zh":"选择推理数据、训练、测试时计算或智能体设计的读者。","confidence":"high","one_line":["A TMLR survey that maps reasoning by when it is achieved and which system components perform it.","按实现阶段和系统架构组织大模型推理前沿的 TMLR 2025 综述。"],"why":"It prevents data, training, inference, and agent design from being compared as if they were the same intervention.","primary_link":"https://mlanthology.org/tmlr/2025/ke2025tmlr-survey/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://llm-reasoning-ai.github.io/"}],"link_count":4,"sections":9},{"id":"multimodal-math-reasoning-survey-2025","title":"A Survey of Mathematical Reasoning in the Era of Multimodal Large Language Model: Benchmark, Method & Challenges","year":2025,"venue":"Findings of ACL 2025","authors":["Yibo Yan","Jiamin Su","Jianxiang He","Fangteng Fu","Xu Zheng","Yuanhuiyi Lyu","Kun Wang","Shen Wang","Qingsong Wen","Xuming Hu"],"authors_zh":"Yibo Yan 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["mathematical-reasoning","multimodal-reasoning","benchmarks"],"tags":["foundations-and-primers","mathematical-reasoning","multimodal-reasoning","acl-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"多模态数学推理综述","best_for_zh":"设计或评测含图像、图形等非纯文本数学推理任务的读者。","confidence":"high","one_line":["A Findings ACL 2025 survey of multimodal mathematical reasoning benchmarks, methods, and challenges.","系统整理多模态大模型数学推理的基准、方法与挑战。"],"why":"It separates failures in multimodal mathematical reasoning into evaluation, methodology, and challenge dimensions.","primary_link":"https://aclanthology.org/2025.findings-acl.614/","links":[],"link_count":2,"sections":9},{"id":"multilingual-reasoning-survey-2025","title":"A Survey of Multilingual Reasoning in Language Models","year":2025,"venue":"Findings of EMNLP 2025","authors":["Akash Ghosh","Debayan Datta","Sriparna Saha","Chirag Agarwal"],"authors_zh":"Akash Ghosh 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["multilingual-reasoning","reasoning-data","benchmarks"],"tags":["foundations-and-primers","multilingual-reasoning","emnlp-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"多语言推理综述","best_for_zh":"想比较不同语言中的推理训练资源、方法与评测结果的读者。","confidence":"high","one_line":["A Findings EMNLP 2025 survey of multilingual reasoning methods, training resources, and evaluation benchmarks.","系统梳理语言模型跨语言推理的方法、数据资源与评测基准。"],"why":"It makes language coverage a concrete part of reasoning-data and benchmark design rather than an afterthought.","primary_link":"https://aclanthology.org/2025.findings-emnlp.474/","links":[],"link_count":2,"sections":9},{"id":"rag-reasoning-systems-survey-2025","title":"A Survey of RAG-Reasoning Systems in Large Language Models","year":2025,"venue":"Findings of EMNLP 2025","authors":["Yangning Li","Weizhi Zhang","Yuyao Yang","Wei-Chieh Huang","Yaozu Wu","Junyu Luo","Yuanchen Bei","Henry Peng Zou","Xiao Luo","Yusheng Zhao","Chunkit Chan","Yankai Chen","Zhongfen Deng","Yinghui Li","Hai-Tao Zheng","Dongyuan Li","Renhe Jiang","Ming Zhang","Yangqiu Song","Philip S. Yu"],"authors_zh":"Yangning Li 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["trace_writing","search_substrate"],"domains":["rag","reasoning","retrieval","agents","benchmarks"],"tags":["foundations-and-primers","rag","reasoning","emnlp-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"RAG 与推理系统综述","best_for_zh":"需要让模型一边检索证据、一边完成复杂推理的读者。","confidence":"high","one_line":["A 2025 EMNLP Findings survey that maps how retrieval and multi-step reasoning can strengthen each other.","用统一框架梳理检索增强生成与多步推理如何相互促进。"],"why":"It separates using reasoning to improve retrieval from using retrieval to supply missing premises for reasoning.","primary_link":"https://aclanthology.org/2025.findings-emnlp.648/","links":[],"link_count":2,"sections":9},{"id":"reasoning-foundation-models-survey-2025","title":"A Survey of Reasoning with Foundation Models: Concepts, Methodologies, and Outlook","year":2025,"venue":"ACM Computing Surveys","authors":["Jiankai Sun","Chuanyang Zheng","Enze Xie","Zhengying Liu","Ruihang Chu","Jianing Qiu","Jiaqi Xu","Mingyu Ding","Hongyang Li","Mengzhe Geng","Yue Wu","Wenhai Wang","Junsong Chen","Zhangyue Yin","Xiaozhe Ren","Jie Fu","Junxian He","Yuan Wu","Qi Liu","Xihui Liu","Yu Li","Hao Dong","Yu Cheng","Ming Zhang","Pheng Ann Heng","Jifeng Dai","Ping Luo","Jingdong Wang","Ji-Rong Wen","Xipeng Qiu","Yike Guo","Hui Xiong","Qun Liu","Zhenguo Li"],"authors_zh":"Jiankai Sun 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","foundation-models","reasoning-data","benchmarks","evaluation"],"tags":["foundations-and-primers","reasoning","survey","acm-computing-surveys",2025],"status":"verified","priority":"必读","paper_type_zh":"基础模型推理综述","best_for_zh":"需要先建立推理数据、方法与评测全景图的读者。","confidence":"high","one_line":["A 2025 survey that connects reasoning tasks to the models, data, methods, and benchmarks used to study them.","把推理任务、模型、方法与基准放进同一张图的 2025 综述。"],"why":"It helps readers separate a reasoning claim from the particular prompt, data, model, and score behind that claim.","primary_link":"https://dl.acm.org/doi/10.1145/3729218","links":[],"link_count":3,"sections":9},{"id":"rl-large-reasoning-models-survey-2025","title":"A Survey of Reinforcement Learning for Large Reasoning Models","year":2025,"venue":"arXiv preprint","authors":["Kaiyan Zhang","Yuxin Zuo","Bingxiang He","Youbang Sun","Runze Liu","Che Jiang","Yuchen Fan","Kai Tian","Guoli Jia","Pengfei Li","Yu Fu","Xingtai Lv","Yuchen Zhang","Sihang Zeng","Shang Qu","Haozhan Li","Shijie Wang","Yuru Wang","Xinwei Long","Fangfu Liu","Xiang Xu","Jiaze Ma","Xuekai Zhu","Ermo Hua","Yihao Liu","Zonglin Li","Huayu Chen","Xiaoye Qu","Yafu Li","Weize Chen","Zhenzhao Yuan","Junqi Gao","Dong Li","Zhiyuan Ma","Ganqu Cui","Zhiyuan Liu","Biqing Qi","Ning Ding","Bowen Zhou"],"authors_zh":"Kaiyan Zhang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["reward_modeling","rlvr","evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","reinforcement-learning","reasoning-data","reward-models","verifiers"],"tags":["foundations-and-primers","reinforcement-learning","reasoning-data","tginghua","survey"],"status":"verified","priority":"必读","paper_type_zh":"大推理模型强化学习综述","best_for_zh":"选择推理训练数据、验证信号或 RL 后训练设计的读者。","confidence":"high","one_line":["A TsinghuaC3I survey of the reinforcement-learning machinery that turns language models into reasoning models.","系统整理大推理模型强化学习、训练资源、奖励与应用的清华综述。"],"why":"It makes reward construction, rollouts, optimization, and evaluation visible as separate data decisions.","primary_link":"https://arxiv.org/abs/2509.08827","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TsinghuaC3I/Awesome-RL-for-LRMs"}],"link_count":3,"sections":9},{"id":"rlhf-survey-2025","title":"A Survey of Reinforcement Learning from Human Feedback","year":2025,"venue":"TMLR 2025","authors":["Timo Kaufmann","Paul Weng","Viktor Bengs","Eyke Hüllermeier"],"authors_zh":"Timo Kaufmann, Paul Weng, Viktor Bengs, Eyke Hüllermeier","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["large-language-models","field-surveys"],"tags":["field-surveys","arxiv-2312.14925","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"跨领域 RLHF 综述与审计框架","best_for_zh":"适合在阅读或设计具体 RLHF 系统前定位反馈、reward model 和策略假设的研究者。","confidence":"high","one_line":["This survey separates RLHF into feedback acquisition, reward learning, and policy learning across control, robotics, and LLMs.","这篇综述把跨控制、机器人和 LLM 的 RLHF 拆成反馈采集、奖励学习与策略学习三阶段。"],"why":"It reveals which data and assumptions support each stage instead of treating RLHF as one opaque operation.","primary_link":"https://arxiv.org/abs/2312.14925","links":[],"link_count":2,"sections":9},{"id":"efficient-llm-training-data-centric-survey-2025","title":"A Survey on Efficient Large Language Model Training: From Data-centric Perspectives","year":2025,"venue":"ACL 2025 (Long Papers)","authors":["Junyu Luo","Bohan Wu","Xiao Luo","Zhiping Xiao","Yiqiao Jin","Rong-Cheng Tu","Nan Yin","Yifan Wang","Jingyang Yuan","Wei Ju","Ming Zhang"],"authors_zh":"Junyu Luo 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":["post-training","data-centric-training","instruction-tuning","reasoning-data"],"tags":["foundations-and-primers","data-centric-training","post-training-survey","data-quality","acl-2025"],"status":"verified","priority":"必读","paper_type_zh":"数据中心的大模型后训练综述","best_for_zh":"关于数据质量、教师生成、筛选和训练预算的数据中心训练综述。","confidence":"high","one_line":["A 2025 ACL survey that treats data selection, quality, synthesis, compression, and self-evolution as distinct levers in efficient post-training.","一篇把数据选择、质量增强、合成、蒸馏压缩和自进化分开讨论的 2025 ACL 后训练综述。"],"why":"It gives Track0 readers a vocabulary for separating a better dataset from a better sampler, teacher, filter, or training budget.","primary_link":"https://aclanthology.org/2025.acl-long.1493/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/luo-junyu/Awesome-Data-Efficient-LLM"}],"link_count":3,"sections":9},{"id":"survey-on-evaluation-of-llm-based-agents-2025","title":"A Survey on Evaluation of LLM-based Agents","year":2025,"venue":"Findings of ACL 2026","authors":["Asaf Yehudai","Lilach Eden","Alan Li","Guy Uziel","Yilun Zhao","Roy Bar-Haim","Arman Cohan","Michal Shmueli-Scheuer"],"authors_zh":"Asaf Yehudai, Lilach Eden, Alan Li, Guy Uziel, Yilun Zhao, Roy Bar-Haim, Arman Cohan, Michal Shmueli-Scheuer","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["agents","evaluation"],"tags":["survey","agent-evaluation","tool-use"],"status":"verified","priority":"必读","paper_type_zh":"LLM 智能体评测综述与资源索引","best_for_zh":"设计、复核或比较网页、软件工程、工具调用与通用智能体评测的研究者。","confidence":"high","one_line":["Surveys LLM-agent evaluation across core capabilities, application domains, benchmark dimensions, and development-time evaluation frameworks.","从能力、应用、基准维度与开发框架四个层面梳理 LLM 智能体评测。"],"why":"It provides a structured way to distinguish evaluation data, environment dynamics, interfaces, metrics, and safety before comparing agent benchmarks.","primary_link":"https://aclanthology.org/2026.findings-acl.1330/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Asaf-Yehudai/LLM-Agent-Evaluation-Survey"}],"link_count":4,"sections":9},{"id":"survey-post-training-llms-2025","title":"A Survey on Post-training of Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Guiyao Tie","Zeli Zhao","Dingjie Song","Fuyang Wei","Rong Zhou","Yurou Dai","Wen Yin","Zhejian Yang","Jiangyue Yan","Yao Su","Zhenhan Dai","Yifeng Xie","Yihan Cao","Lichao Sun","Pan Zhou","Lifang He","Hechang Chen","Yu Zhang","Qingsong Wen","Tianming Liu","Neil Zhenqiang Gong","Jiliang Tang","Caiming Xiong","Heng Ji","Philip S. Yu","Jianfeng Gao"],"authors_zh":"Guiyao Tie 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":["post-training","reasoning-data","alignment"],"tags":["post-training-survey","taxonomy","alignment-data","reasoning-data"],"status":"verified","priority":"必读","paper_type_zh":"后训练与大推理模型综述","best_for_zh":"整理微调、对齐、推理、效率与整合适配的后训练综述。","confidence":"high","one_line":["A 2025 map of post-training that connects data, feedback, reasoning, and system stages before readers enter specialized tracks.","一篇把后训练中的数据、反馈、推理与系统阶段连成概念地图的 2025 综述。"],"why":"It makes the provenance and role of a post-training signal an explicit routing question rather than treating all fine-tuning data alike.","primary_link":"https://arxiv.org/abs/2503.06072","links":[],"link_count":2,"sections":9},{"id":"systematic-base-model-rm-2025","title":"A Systematic Analysis of Base Model Choice for Reward Modeling","year":2025,"venue":"EMNLP 2025","authors":["Kian Ahrabian","Pegah Jandaghi","Negar Mokhberian","Sai Praneeth Karimireddy","Jay Pujara"],"authors_zh":"Kian Ahrabian，Pegah Jandaghi，Negar Mokhberian，Sai Praneeth Karimireddy，Jay Pujara。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"奖励模型基座选择与可靠性审计","best_for_zh":"训练、筛选或审计偏好奖励模型的研究者。","confidence":"medium","one_line":["Accepted analysis of base-model choices as a reward-model reliability variable.","系统证明 RM 基座选择与训练阶段会显著改变 RewardBench 表现，并给出低维筛选线索。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://aclanthology.org/2025.emnlp-main.8/","links":[],"link_count":2,"sections":9},{"id":"systematic-preference-rs-mcts-2025","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","year":2025,"venue":"NAACL 2025","authors":["Joongwon Kim","Anirudh Goyal","Aston Zhang","Bo Xiong","Rui Hou","Melanie Kambadur","Dhruv Mahajan","Hannaneh Hajishirzi","Liang Tan"],"authors_zh":"Joongwon Kim、Anirudh Goyal、Aston Zhang、Bo Xiong、Rui Hou、Melanie Kambadur、Dhruv Mahajan、Hannaneh Hajishirzi、Liang Tan","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","pairwise_preference"],"training_use":["preference_learning","sft","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["instruction_following","constrained_generation"],"tags":["preference-data","rejection-sampling","mcts","shared-prefix","response-contrast","instruction-following","verifiable-constraints","dpo","data-ablation","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"可验证偏好数据构造与消融研究","best_for_zh":"研究偏好数据、拒绝采样、MCTS与指令遵循验证的读者","confidence":"medium","one_line":["A 47,198-prompt synthetic constraint suite is used to compare independent RS pairs with shared-prefix MCTS pairs and isolate the effects of pair contrast and prompt difficulty on DPO.","论文以47,198个合成约束提示为控制平台，对比独立拒绝采样与共享前缀MCTS偏好对，分析对比度和提示难度对DPO的影响。"],"why":"It converts preference-data design choices into controlled experimental variables, while the absence of released prompts, pairs, verifiers, trees, and rejects demonstrates the gap between a documented recipe and an auditable open release.","primary_link":"https://aclanthology.org/2025.naacl-long.552/","links":[],"link_count":4,"sections":9},{"id":"rpc-confidence-sampling-2025","title":"A Theoretical Study on Bridging Internal Probability and Self-Consistency for LLM Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Zhi Zhou","Yuhao Tan","Zenan Li","Yuan Yao","Lan-Zhe Guo","Yu-Feng Li","Xiaoxing Ma"],"authors_zh":"Zhi Zhou、Yuhao Tan、Zenan Li、Yuan Yao、Lan-Zhe Guo、Yu-Feng Li、Xiaoxing Ma（机构：南京大学软件新技术国家重点实验室、苏黎世联邦理工学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code-reasoning","general-reasoning"],"tags":["test-time-compute","self-consistency","confidence-estimation","reasoning-pruning","sampling"],"status":"verified","priority":"必读","paper_type_zh":"测试时采样理论与高效推理研究","best_for_zh":"在有限推理采样数下设计置信度候选选择的读者。","confidence":"high","one_line":["RPC combines internal probability with self-consistency and prunes low-probability paths to reduce sampling cost in reasoning.","RPC 将内部概率与自一致性结合，并裁剪低概率路径以降低推理采样成本。"],"why":"It gives both a theory and an implementation for spending fewer inference samples without reverting to a pure perplexity selector.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/7e9afa9a02857bce4515247842471444-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WNJXYK/RPC"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/WNJXYK/neurips-2025-rpc-resources"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/spaces/WNJXYK/RPC"},{"key":"project","label":["Project","项目主页"],"url":"https://zhouz.dev/RPC/"}],"link_count":6,"sections":9},{"id":"a-star-thought-2025","title":"A*-Thought: Efficient Reasoning via Bidirectional Compression for Low-Resource Settings","year":2025,"venue":"NeurIPS 2025","authors":["Xiaoang Xu","Shuo Wang","Xu Han","Zhenghao Liu","Huijia Wu","Peipei Li","Zhiyuan Liu","Maosong Sun","Zhaofeng He"],"authors_zh":"Xiaoang Xu、Shuo Wang、Xu Han、Zhenghao Liu、Huijia Wu、Peipei Li、Zhiyuan Liu、Maosong Sun、Zhaofeng He","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["mathematics"],"tags":["a-star-thought","chain-of-thought-compression","bidirectional-importance","a-star-search","efficient-reasoning","sft"],"status":"partial","priority":"必读","paper_type_zh":"推理轨迹压缩数据集与搜索构造方法","best_for_zh":"关注长链推理压缩、SFT 数据和验证器边界的读者","confidence":"high","one_line":["A*-Thought compresses s1K-1.1 long-CoT traces with BIS-guided A* search and s1.1-32B validation, releasing a one-thousand-record SFT dataset and model checkpoints.","A*-Thought 使用 BIS 引导的 A* 搜索与 s1.1-32B 验证，将 s1K-1.1 长链推理压缩为一千条 SFT 轨迹。"],"why":"It makes reasoning-token reduction a reusable trace-construction problem while preserving model-based path validation and upstream provenance as audit boundaries.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/9ac5238b5c033fc49c2932a50df2bd47-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AI9Stars/AStar-Thought"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xxang/AStar-Thought-1k"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/xxang/astar-thought"}],"link_count":6,"sections":9},{"id":"absolute-zero-2025","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","year":2025,"venue":"NeurIPS 2025 (Main Conference Track)","authors":["Andrew Zhao","Yiran Wu","Yang Yue","Tong Wu","Quentin Xu","Matthieu Lin","Shenzhi Wang","Qingyun Wu","Zilong Zheng","Gao Huang"],"authors_zh":"Andrew Zhao、Yiran Wu、Yang Yue、Tong Wu、Quentin Xu、Matthieu Lin、Shenzhi Wang、Qingyun Wu、Zilong Zheng、Gao Huang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr"],"construction_layer":["self_play_anchor","reward_verifier_layer","optimizer_scaffold"],"domains":["code_reasoning","mathematical_reasoning","procedural_task_generation","self_play"],"tags":["absolute-zero","self-play","rlvr","procedural-task-generation","code-execution","autotelic-curriculum","trr-plus-plus","zero-data-boundary"],"status":"partial","priority":"必读","paper_type_zh":"自生成程序任务、可执行验证与在线 RLVR 构造 recipe","best_for_zh":"研究 self-play curriculum、程序执行 verifier、无外部任务数据 RLVR，或审计自动生成任务中的安全、捷径与 lineage 风险的读者","confidence":"high","one_line":["Absolute Zero turns a growing buffer of self-proposed Python triplets into an online RLVR curriculum, rewarding proposer learnability from eight solver attempts and solver correctness by execution, while releasing the recipe and seeds but not the complete paper-run trajectories.","Absolute Zero 让同一 policy 在线提出并求解 Python deduction、abduction 与 induction 任务，以执行结果和八次 solver 尝试构造 RLVR reward；其“zero data”仅指 RL 阶段无外部人工或蒸馏任务答案数据，完整 rollout 并未发布。"],"why":"It makes a zero-external-task-data self-play loop operational and auditable, but also exposes the boundary that \"zero data\" still relies on pretrained models, human-designed prompts and verifiers, and can retain shortcut-bearing programs whose full lineage is not released.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/file/9837dc00ff67d176373268ed48042d49-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LeapLabTHU/Absolute-Zero-Reasoner"},{"key":"data","label":["Data","数据"],"url":"https://github.com/LeapLabTHU/Absolute-Zero-Reasoner/tree/paper/data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/andrewzh/absolute-zero-reasoner"},{"key":"project","label":["Project","项目主页"],"url":"https://andrewzh112.github.io/absolute-zero-reasoner/"}],"link_count":11,"sections":9},{"id":"abstentionbench-2025","title":"AbstentionBench","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Polina Kirichenko","Mark Ibrahim","Kamalika Chaudhuri","Samuel J. Bell"],"authors_zh":"Polina Kirichenko、Mark Ibrahim、Kamalika Chaudhuri、Samuel J. Bell","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","audit_failure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","safety_alignment"],"construction_layer":["release_audit"],"domains":["abstention","factuality","uncertainty"],"tags":[],"status":"verified","priority":"必读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["AbstentionBench evaluates whether LLMs know when not to answer across unknown, underspecified, false-premise, subjective, and stale-information questions.","检验模型面对未知、歧义和过时问题时能否正确拒答，把“知道何时不答”纳入推理可靠性审计。"],"why":"It is a direct audit surface for reasoning models: stronger reasoning can still fail if the model confidently answers unanswerable questions instead of abstaining.","primary_link":"https://arxiv.org/abs/2506.09038","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/AbstentionBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/AbstentionBench"}],"link_count":4,"sections":9},{"id":"speculative-search-2025","title":"Accelerating Large Language Model Reasoning via Speculative Search","year":2025,"venue":"ICML 2025","authors":["Zhihai Wang","Jie Wang","Jilai Pan","Xilin Xia","Huiling Zhen","Mingxuan Yuan","Jianye Hao","Feng Wu"],"authors_zh":"Zhihai Wang、Jie Wang、Jilai Pan、Xilin Xia、Huiling Zhen、Mingxuan Yuan、Jianye Hao、Feng Wu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["mathematics","reasoning"],"tags":["speculative-search","test-time-search","process-reward-model","thought-generation","trace-selection","inference-acceleration"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["An ICML 2025 tree-search accelerator that accepts PRM-scored small-model thoughts above a history-based threshold and replaces rejected thoughts through lossless large-model speculative decoding.","这项 ICML 2025 工作是一种树搜索加速器，接受过程奖励模型打分高于历史阈值的小模型思路，并用无损的大模型推测解码替换被拒绝的思路。"],"why":"For rollout and test-time trace curation, it exposes the thought, score, dynamic threshold, accept-or-correct decision, and search state that should be retained, while the official release stops at code rather than auditable trace data.","primary_link":"https://proceedings.mlr.press/v267/wang25di.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MIRALab-USTC/LLMReasoning-SpecSearch"}],"link_count":4,"sections":9},{"id":"acemath-2025","title":"AceMath: Advancing Frontier Math Reasoning with Post-Training and Reward Modeling","year":2025,"venue":"Findings of ACL 2025","authors":["Zihan Liu","Yang Chen","Mohammad Shoeybi","Bryan Catanzaro","Wei Ping"],"authors_zh":"Zihan Liu、Yang Chen、Mohammad Shoeybi、Bryan Catanzaro、Wei Ping","tracks":["instruction_demonstration_rationale_data"],"source_role":["model_report","data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","code","general-instruction-following"],"tags":["instruction-demonstration-rationale","math-reasoning","staged-sft","synthetic-data","cross-model-filtering","arxiv-2412.15084","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"分阶段数学指令数据发布与后训练研究","best_for_zh":"适合构建大规模数学监督微调混合，或比较题目演化、答案交叉核对与分阶段课程。","confidence":"high","one_line":["AceMath releases 2.26M, 1.63M, and 1.66M-row SFT stages for general-to-math post-training with explicit synthesis and filtering controls.","AceMath 发布三阶段通用到数学监督微调数据，以统一教师解答、跨模型答案核对和数据筛选支撑后训练。"],"why":"It makes a strong general-before-math SFT curriculum and its synthetic-data quality decisions reusable instead of reporting only the final models.","primary_link":"https://aclanthology.org/2025.findings-acl.206/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/AceMath-Instruct-Training-Data"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/adlr/acemath/"}],"link_count":6,"sections":9},{"id":"acereason-nemotron-1-1-2025","title":"AceReason-Nemotron 1.1: Advancing Math and Code Reasoning through SFT and RL Synergy","year":2025,"venue":"arXiv preprint","authors":["Zihan Liu","Zhuolin Yang","Yang Chen","Chankyu Lee","Mohammad Shoeybi","Bryan Catanzaro","Wei Ping"],"authors_zh":"Zihan Liu、Zhuolin Yang、Yang Chen、Chankyu Lee、Mohammad Shoeybi、Bryan Catanzaro、Wei Ping","tracks":["frontier_reports_data_disclosure_ledger","instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["model_report","data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","code"],"tags":["nvidia","acereason-nemotron","frontier-report","data-disclosure-ledger","sft","rlvr","grpo","math","code","deepseek-r1"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型报告、数据发布与训练配方披露","best_for_zh":"需要审计数学/代码 SFT、RLVR 反馈合同、分阶段课程与发布边界的研究者","confidence":"high","one_line":["AceReason-Nemotron 1.1 releases a four-field, DeepSeek-R1-generated math/code SFT corpus and model weights while reporting staged, rule-verified on-policy GRPO; the consumed SFT subset, complete code-RL tasks/tests, and example-to-checkpoint lineage remain unavailable.","AceReason-Nemotron 1.1 发布四字段、由 DeepSeek-R1 生成的数学/代码 SFT 语料与模型权重，并报告分阶段、规则验证的严格 on-policy GRPO；但实际使用的 SFT 子集、完整代码 RL 任务/测试和样本到检查点谱系仍不可得。"],"why":"For the frontier-report disclosure ledger, it lets readers compare a concrete public SFT object and several RL/verifier ablations with the missing artifacts that still block full reproduction and verifier audit.","primary_link":"https://arxiv.org/abs/2506.13284","links":[{"key":"code","label":["Code","代码"],"url":"https://huggingface.co/nvidia/AceReason-Nemotron-14B/blob/main/README_EVALUATION.md"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/AceReason-1.1-SFT"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/AceReason-Nemotron-1.1-7B"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/acereason"}],"link_count":7,"sections":9},{"id":"acereason-nemotron-advancing-math-and-code-reasoning-through-reinforcement-learning","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","year":2025,"venue":"arXiv","authors":["Yang Chen","Zhuolin Yang","Zihan Liu","Chankyu Lee","Peng Xu","Mohammad Shoeybi","Bryan Catanzaro","Wei Ping"],"authors_zh":"Yang Chen, Zhuolin Yang, Zihan Liu, Chankyu Lee, Peng Xu, Mohammad Shoeybi, Bryan Catanzaro, Wei Ping","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** AceReason-Nemotron studies sequential cross-domain RL with about 49K rule-verifiable math prompts and executable code tasks.","AceReason-Nemotron 用约 49K 可规则验证数学题和执行式代码任务研究顺序跨域 RL。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2505.16400","links":[{"key":"code","label":["Code","代码"],"url":"https://huggingface.co/nvidia/AceReason-Nemotron-14B/blob/main/README_EVALUATION.md"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/AceReason-Math"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/AceReason-Nemotron-14B"}],"link_count":4,"sections":9},{"id":"sams-batchwise-dpo-scheduling-2025","title":"Adaptive Batch-Wise Sample Scheduling for Direct Preference Optimization","year":2025,"venue":"NeurIPS 2025","authors":["Zixuan Huang","Yikun Ban","Lean Fu","Xiaojie Li","Zhongxiang Dai","Jianxin Li","Deqing Wang"],"authors_zh":"Zixuan Huang、Yikun Ban、Lean Fu、Xiaojie Li、Zhongxiang Dai、Jianxin Li、Deqing Wang（北京航空航天大学、字节跳动、香港中文大学（深圳））","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-following"],"tags":["data-selection","dpo","preference-learning","sample-scheduling","alignment"],"status":"verified","priority":"必读","paper_type_zh":"动态偏好样本调度与直接偏好优化研究","best_for_zh":"适合研究偏好数据应在训练过程的何时、以何种顺序被模型使用的读者。","confidence":"high","one_line":["SamS dynamically schedules the preference pairs used by each DPO batch from learning feedback.","SamS 根据模型在每个批次中的学习反馈，动态安排 DPO 实际使用的偏好对。"],"why":"It turns preference-data use from a static pre-filtering decision into a model-state-dependent training decision.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/36ecc1d1b883afc0e882876cbdd123ab-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hzx122/SamS"}],"link_count":5,"sections":9},{"id":"abcd-cyclic-diffusion-tts-2025","title":"Adaptive Inference-Time Scaling via Cyclic Diffusion Search","year":2025,"venue":"NeurIPS 2025 Poster","authors":["Gyubin Lee","Truong Nhat Nguyen Bao","Jaesik Yoon","Dongwoo Lee","Minsu Kim","Yoshua Bengio","Sungjin Ahn"],"authors_zh":"Gyubin Lee、Truong Nhat Nguyen Bao、Dongwoo Lee（KAIST）；Jaesik Yoon（KAIST、SAP）；Minsu Kim（Mila、KAIST）；Yoshua Bengio（Mila、蒙特利尔大学）；Sungjin Ahn（KAIST、NYU）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","diffusion","adaptive-search","stopping"],"status":"verified","priority":"可读","paper_type_zh":"自适应扩散推理扩展研究","best_for_zh":"研究自回归采样以外的测试时预算分配的读者。","confidence":"high","one_line":["ABCD uses cyclic diffusion search to adapt exploration depth and stopping time to the available inference budget.","ABCD 通过循环式双向扩散搜索，自适应决定探索深度和停止时间，从而按推理预算扩展计算。"],"why":"It makes denoising depth and search termination explicit test-time scaling decisions.","primary_link":"https://arxiv.org/abs/2505.14036","links":[{"key":"code","label":["Code","代码"],"url":"https://lee-gyubin.github.io/ABCD_project_page/"}],"link_count":4,"sections":9},{"id":"uapo-uncertainty-utility-anchor-2025","title":"Adaptive Preference Optimization with Uncertainty-aware Utility Anchor","year":2025,"venue":"Findings of EMNLP 2025","authors":["Xiaobo Wang","Zixia Jia","Jiaqi Li","Qi Liu","Zilong Zheng"],"authors_zh":"Xiaobo Wang, Zixia Jia, Jiaqi Li, Qi Liu, Zilong Zheng","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-tuning"],"tags":["preference-optimization","unpaired-data","uncertainty","utility-anchor","data-utilization"],"status":"verified","priority":"可读","paper_type_zh":"不确定性感知的离线偏好优化研究","best_for_zh":"希望复用不平衡或未配对偏好记录、而非因不满足 DPO 成对要求而丢弃它们的读者。","confidence":"high","one_line":["UAPO learns a prompt-conditioned utility anchor so paired and unpaired preference records can be used under an uncertainty-aware objective.","UAPO 学习与提示相关的效用锚点，使成对和未配对的偏好记录都能在不确定性感知目标下被使用。"],"why":"It treats the availability and uncertainty of preference records as part of the objective rather than assuming every usable example already has a clean winner-loser pair.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1046/","links":[],"link_count":2,"sections":9},{"id":"adaptivestep-model-confidence-segmentation-2025","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","year":2025,"venue":"ICML 2025","authors":["Yuliang Liu","Junjie Lu","Zhaoling Chen","Chaofeng Qu","Jason Klein Liu","Chonghan Liu","Zefan Cai","Yunhui Xia","Li Zhao","Jiang Bian","Chuheng Zhang","Wei Shen","Zhouhan Lin"],"authors_zh":"Yuliang Liu、Junjie Lu、Zhaoling Chen、Chaofeng Qu、Jason Klein Liu、Chonghan Liu、Zefan Cai、Yunhui Xia、Li Zhao、Jiang Bian、Chuheng Zhang、Wei Shen、Zhouhan Lin","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","code-generation","process-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要自动分割长推理轨迹、避免固定长度或换行规则，并训练过程奖励模型的研究者。","confidence":"high","one_line":["AdaptiveStep replaces hand-written step delimiters with model-confidence boundaries, producing process-reward supervision that targets actual decision points in math and code traces.","AdaptiveStep 用模型置信度确定推理步骤边界，替代人工规则切分，使数学和代码轨迹的过程奖励对准实际决策点。"],"why":"It makes the granularity of process supervision a learned, inspectable construction decision rather than a formatting artifact.","primary_link":"https://arxiv.org/abs/2502.13943","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Lux0926/ASPRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Lux0926/ASPRM-MATHCODE-DeepSeek-Training-Dataset"}],"link_count":6,"sections":9},{"id":"adaptthink-2025","title":"AdaptThink: Reasoning Models Can Learn When to Think","year":2025,"venue":"EMNLP 2025","authors":["Jiajie Zhang","Nianyi Lin","Lei Hou","Ling Feng","Juanzi Li"],"authors_zh":"Jiajie Zhang、Nianyi Lin、Lei Hou、Ling Feng、Juanzi Li","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["rlvr","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics"],"tags":["adaptthink","thinking-mode-selection","rlvr","efficient-reasoning"],"status":"partial","priority":"必读","paper_type_zh":"测试时推理模式选择与 RLVR 训练配方","best_for_zh":"关注推理 token 分配、Thinking/NoThinking 轨迹选择、答案验证和 rollout 审计边界的读者","confidence":"high","one_line":["Answer-verified RL chooses a full math trace or an immediate answer.","AdaptThink 以答案正确性为终点信号，结合受约束优化与重要性采样，在完整思考和直接回答两种输出模式之间学习按题目难度分配推理。"],"why":"It turns reasoning-token allocation into trace selection and exposes missing rollout evidence.","primary_link":"https://aclanthology.org/2025.emnlp-main.184/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/AdaptThink"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/THU-KEG/adaptthink"}],"link_count":6,"sections":9},{"id":"openai-gpt-5-codex-system-card-2025","title":"Addendum to GPT-5 system card: GPT-5-Codex","year":2025,"venue":"OpenAI system-card addendum","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["agent_training","evaluation","safety_alignment"],"construction_layer":["prompt_sourcing","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["software_engineering","coding_agents","cybersecurity","safety"],"tags":["openai","gpt-5-codex","system-card","coding-agent","real-world-software-engineering","reinforcement-learning","code-review","pull-request-preferences","executable-tests","malware-safety","prompt-injection","human-evaluation","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿编码代理系统卡增补与数据披露账本","best_for_zh":"审计 Codex 家族 SWE 任务、代码审查反馈、评测 verifier 与版本披露差异的研究者","confidence":"high","one_line":["GPT-5-Codex discloses RL on real-world project building, feature and test work, debugging, refactoring, and code review plus synthetic malware and new prompt-injection safety data, but not the task records, preference or reward representation, trajectories, split, or reusable artifacts.","GPT-5-Codex 披露了面向真实项目构建、功能与测试、调试、重构和代码审查的 RL，以及继承的恶意软件安全数据和新 prompt-injection 数据；其价值是区分训练目标、人工/测试评测与部署防护，但 reward 表示、trajectory、split 和可复用 artifact 均未公开。"],"why":"It is a useful coding-agent disclosure ledger because it names concrete training task families and separates code-review training, experienced-engineer evaluation, executable tests, safety data, traffic observations, and product sandboxing that could otherwise be conflated into an unsupported reward or dataset claim.","primary_link":"https://cdn.openai.com/pdf/97cc5669-7a25-4e63-b15f-5fd5bdc4d149/gpt-5-codex-system-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-upgrades-to-codex/"}],"link_count":3,"sections":9},{"id":"openai-gpt-5-2-codex-system-card-2025","title":"Addendum to GPT-5.2 System Card: GPT-5.2-Codex","year":2025,"venue":"OpenAI system-card addendum","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["agent_training","evaluation","safety_alignment","test_time_compute"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["software_engineering","coding_agents","cybersecurity","safety","biosecurity","ai_self_improvement"],"tags":["openai","gpt-5-2-codex","system-card","coding-agent","context-compaction","destructive-action","conflicting-edits","reinforcement-learning","hidden-unit-tests","internal-pr-evaluation","safety-training","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿编码智能体系统卡补充报告与数据披露台账","best_for_zh":"需要区分编码智能体训练数据、隐藏测试评测、产品 sandbox 与安全部署证据的研究者和审计者","confidence":"high","one_line":["The GPT-5.2-Codex addendum discloses RL rollouts where a user model makes conflicting edits and preserving them earns positive reinforcement, while separately documenting internal PR evaluations with human-written prompts/tests/hints and hidden unit tests but withholding the underlying training records and reward implementation.","GPT-5.2-Codex 补充系统卡披露了一项编码智能体 RL 干预：user model 在 rollout 中制造冲突编辑，模型保留用户修改可获得正向强化；但训练记录、奖励实现和逐项 lineage 均未公开。"],"why":"It is a high-value frontier disclosure ledger because it exposes one concrete state-changing coding-agent RL intervention and multiple executable evaluation contracts while making the train/eval boundary, missing lineage, and non-reproducibility explicit.","primary_link":"https://cdn.openai.com/pdf/ac7c37ae-7f4c-4442-b741-2eabdeaf77e0/oai_5_2_Codex.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://deploymentsafety.openai.com/gpt-5-2-codex"}],"link_count":2,"sections":9},{"id":"openai-o3-o4-mini-codex-addendum-2025","title":"Addendum to OpenAI o3 and o4-mini System Card: Codex","year":2025,"venue":"OpenAI system-card addendum","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["state_action_level","scalar_reward"],"training_use":["agent_training","safety_alignment","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["coding","software_engineering","agentic_tool_use"],"tags":["openai","o3","o4-mini","codex","frontier-report","data-disclosure-ledger","coding-agent","reinforcement-learning","safety-training","environment-perturbation","prompt-injection"],"status":"partial","priority":"必读","paper_type_zh":"编码智能体系统卡补充报告","best_for_zh":"审计闭源编码智能体训练和安全披露边界的读者","confidence":"high","one_line":["OpenAI's Codex addendum reports RL on real-world coding tasks plus synthetic/perturbed safety environments for malware, false-completion, and prompt-injection risks, but does not release task manifests, trajectories, the main reward, or training artifacts.","OpenAI 的 Codex addendum 报告了现实编码任务上的 RL、合成安全管线和窄动作一致性目标，但未披露任务、奖励、来源和轨迹账本。"],"why":"It is a useful frontier disclosure ledger because it names concrete safety data objects and an action-consistency reward while showing that strong evaluation scores and sandbox mitigations do not make the underlying coding-agent RL pipeline auditable.","primary_link":"https://cdn.openai.com/pdf/8df7697b-c1b2-4222-be00-1fd3298f351d/codex_system_card.pdf","links":[],"link_count":2,"sections":9},{"id":"advancedif-instruction-following-2025","title":"AdvancedIF: Rubric-Based Benchmarking and Reinforcement Learning for Advancing LLM Instruction Following","year":2025,"venue":"ACL 2026","authors":["Yun He","Wenzhe Li","Hejia Zhang","Songlin Li","Karishma Mandyam","Sopan Khosla","Yuanhao Xiong","Nanshu Wang","Xiaoliang Peng","Beibin Li","Shengjie Bi","Shishir G. Patil","Qi Qi","Shengyu Feng","Julian Katz-Samuels","Richard Yuanzhe Pang","Sujan Gonugondla","Hunter Lang","Yue Yu","Yundi Qian","Maryam Fazel-Zarandi","Licheng Yu","Amine Benhalloum","Hany Awadalla","Manaal Faruqui"],"authors_zh":"Yun He, Wenzhe Li, Hejia Zhang, Songlin Li, Karishma Mandyam 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","sft","reward_modeling","rlvr","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["instruction_following","alignment"],"tags":["instruction_following","rubric","reward_model","rl"],"status":"verified","priority":"必读","paper_type_zh":"高级指令跟随的专家 rubric 与奖励学习基准","best_for_zh":"需要训练或评估指令跟随 Judge、奖励模型和 RL 系统的研究者。","confidence":"high","one_line":["AdvancedIF uses human-written rubrics to evaluate and train complex, multi-turn, and system-level instruction following.","AdvancedIF 将专家 rubric 用于复杂、多轮和系统提示指令跟随的评测与强化学习。"],"why":"It turns hard-to-verify instruction following into interpretable criterion-level rewards.","primary_link":"https://arxiv.org/abs/2511.10507","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/AdvancedIF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/AdvancedIF"}],"link_count":5,"sections":9},{"id":"ultrainteract-preference-trees-2025","title":"Advancing LLM Reasoning Generalists with Preference Trees","year":2025,"venue":"ICLR 2025","authors":["Lifan Yuan","Ganqu Cui","Hanbin Wang","Ning Ding","Xingyao Wang","Boji Shan","Zeyuan Liu","Jia Deng","Huimin Chen","Ruobing Xie","Yankai Lin","Zhenghao Liu","Bowen Zhou","Hao Peng","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Lifan Yuan, Ganqu Cui, Hanbin Wang, Ning Ding, Xingyao Wang, Boji Shan, Zeyuan Liu, Jia Deng, Huimin Chen, Ruobing Xie, Yankai Lin, Zhenghao Liu, Bowen Zhou, Hao Peng, Zhiyuan Liu, Maosong Sun","tracks":["instruction_demonstration_rationale_data","preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level","pairwise_preference"],"training_use":["sft","preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["mathematical-reasoning","code-generation","logical-reasoning","tool-use"],"tags":["reasoning-demonstrations","preference-trees","tool-augmented-rationales","multi-turn-correction","alignment-data"],"status":"verified","priority":"必读","paper_type_zh":"推理示范数据集与偏好树构造配方","best_for_zh":"适合构建或审计数学、代码与逻辑推理 SFT 数据的读者。","confidence":"high","one_line":["UltraInteract turns objectively checked reasoning actions and correction trajectories into public SFT demonstrations and companion preference pairs.","UltraInteract 把经过客观检查的推理动作与纠错轨迹转成公开的 SFT 示范，并同时发布偏好对。"],"why":"It exposes how instructions, actor-generated rationales, execution feedback, critiques, and correctness decisions become reusable reasoning post-training records.","primary_link":"https://openreview.net/forum?id=2ea5TNVR0c","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/Eurus"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openbmb/UltraInteract_sft"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/openbmb/eurus-660bc40bec5376b3adc9d1c5"}],"link_count":6,"sections":9},{"id":"advancing-math-data-synthesis-2025","title":"Advancing Mathematical Reasoning in Language Models: The Impact of Problem-Solving Data, Data Synthesis Methods, and Training Stages","year":2025,"venue":"ICLR 2025","authors":["Zui Chen","Tianqiao Liu","Mi Tian","Qing Tong","Weiqi Luo","Zitao Liu"],"authors_zh":"Zui Chen、Tianqiao Liu、Mi Tian、Qing Tong、Weiqi Luo、Zitao Liu","tracks":["instruction_demonstration_rationale_data","data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study","model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["distillation","sft"],"construction_layer":["prompt_sourcing","trace_writing","scaling_report"],"domains":["mathematics"],"tags":["mathgpt","math-data-synthesis","response-diversification","query-expansion","retrospective-enhancement","tutorship-amplification","continued-pretraining","sft"],"status":"partial","priority":"可读","paper_type_zh":"数学数据合成算子与训练阶段对照研究","best_for_zh":"关注数学推理数据合成、教师蒸馏、CPT/SFT 选择、污染与发布审计的研究者","confidence":"medium","one_line":["Compares four math problem-synthesis operators and controlled CPT/SFT allocations, but releases MathGPT-8B rather than the source or synthetic corpora.","比较 response diversification、query expansion、retrospective enhancement 与 tutorship amplification 四种数学数据变换，并系统对照相同问题求解数据在 CPT 与 SFT 阶段的作用。"],"why":"Treats transformation semantics and training stage as separate reasoning-data choices and identifies provenance, judge, rejection, budget, and license records still required for audit or reuse.","primary_link":"https://openreview.net/forum?id=GtpubstM1D","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/ai4ed/MathGPT-8B"}],"link_count":4,"sections":9},{"id":"aprm-adversarial-process-reward-2025","title":"Adversarial Training for Process Reward Models","year":2025,"venue":"arXiv preprint under review","authors":["Gurusha Juneja","Deepak Nathani","William Yang Wang"],"authors_zh":"Gurusha Juneja、Deepak Nathani、William Yang Wang（加州大学圣塔芭芭拉分校）","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["process_reward"],"training_use":["process_supervision","test_time_compute"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["process-reward-model","adversarial-training","hard-negatives","process-labels"],"status":"verified","priority":"必读","paper_type_zh":"对抗式过程监督构造与过程奖励模型训练研究","best_for_zh":"适合设计步骤级奖励模型自适应负样本课程的读者。","confidence":"medium","one_line":["APRM trains a generator to create PRM-deceiving wrong steps and continuously feeds those hard negatives back into process reward modeling.","APRM 训练生成器制造能欺骗过程奖励模型的错误步骤，并持续把这些难负样本反馈给过程奖励模型训练。"],"why":"It changes the source and hardness distribution of negative process labels as the reward model improves.","primary_link":"https://arxiv.org/abs/2511.22888","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/gurusha01/PRM_NIPS"},{"key":"project","label":["Project","项目主页"],"url":"https://gurusha01.github.io/PRM_NIPS/"}],"link_count":4,"sections":9},{"id":"agentgym-2025","title":"AGENT GYM: Evaluating and Training Large Language Model-based Agents across Diverse Environments","year":2025,"venue":"ACL 2025 Long Papers","authors":["Zhiheng Xi","Yiwen Ding","Wenxiang Chen","Boyang Hong","Honglin Guo","Junzhe Wang","Xin Guo","Dingwen Yang","Chenyang Liao","Wei He","Songyang Gao","Lu Chen","Rui Zheng","Yicheng Zou","Tao Gui","Qi Zhang","Xipeng Qiu","Xuanjing Huang","Zuxuan Wu","Yu-Gang Jiang"],"authors_zh":"Zhiheng Xi, Yiwen Ding, Wenxiang Chen, Boyang Hong, Honglin Guo, Junzhe Wang, Xin Guo, Dingwen Yang, Chenyang Liao, Wei He, Songyang Gao, Lu Chen, Rui Zheng, Yicheng Zou, Tao Gui, Qi Zhang, Xipeng Qiu, Xuanjing Huang, Zuxuan Wu, Yu-Gang Jiang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["environmental","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent-reasoning","web","embodied-ai","games","tool-use","programming"],"tags":["instruction-demonstration-rationale","agent-trajectories","environment-reward","agentgym","arxiv-2406.04151"],"status":"verified","priority":"必读","paper_type_zh":"跨环境代理轨迹数据集与交互训练框架研究","best_for_zh":"适合构建可执行代理语料、比较环境奖励和研究跨环境监督训练的读者。","confidence":"high","one_line":["AgentTraj-L releases 14,485 reward-filtered ReAct demonstrations from 11 interactive environments for cross-environment agent SFT.","AgentTraj-L 公开来自十一个交互环境的 14,485 条奖励过滤示范，用于跨环境代理监督微调。"],"why":"It pairs static trainable traces with standardized live environments, making reward provenance and cross-environment comparisons unusually explicit.","primary_link":"https://aclanthology.org/2025.acl-long.1355/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WooooDyy/AgentGym"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AgentGym/AgentTraj-L"},{"key":"project","label":["Project","项目主页"],"url":"https://agentgym.github.io/"}],"link_count":7,"sections":9},{"id":"agent-lightning-2025","title":"Agent Lightning: Train ANY AI Agents with Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["Xufang Luo","Yuge Zhang","Zhiyuan He","Zilong Wang","Siyun Zhao","Dongsheng Li","Luna K. Qiu","Yuqing Yang"],"authors_zh":"Xufang Luo、Yuge Zhang、Zhiyuan He、Zilong Wang、Siyun Zhao、Dongsheng Li、Luna K. Qiu、Yuqing Yang","tracks":["frontier_reports_data_disclosure_ledger","environment_agent_trajectory_data","training_usage_optimization_objectives"],"source_role":["construction_recipe","verifier_reward","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["agent_training","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_training","text_to_sql","retrieval_augmented_generation","mathematics","tool_use","multi_agent_systems"],"tags":["agent-lightning","microsoft","agent-optimization","reinforcement-learning","transition-schema","observability","opentelemetry","lightningrl","trace-disclosure","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"智能体强化学习框架、可观测性轨迹接口与数据披露边界审计","best_for_zh":"关注智能体后训练、状态动作转移、OpenTelemetry 追踪、奖励设计、用户数据治理和隐私的研究者与工程人员","confidence":"high","one_line":["Microsoft's MIT framework converts user-run agent executions into state/action/reward transitions and trace spans for LightningRL, but its official release is framework code and examples—not a canonical corpus, reward ledger, environment bundle or benchmark data release.","Agent Lightning 是微软的 MIT 智能体优化框架，它将用户运行的智能体执行转换为 LightningRL 所需的状态/动作/奖励转移和 trace span；官方发布的是框架代码与示例，不是规范语料、奖励账本、环境包或基准数据发布。"],"why":"It exposes a reusable transition and observability contract for training agents while sharply separating framework-level trace capture from the missing provenance, rights, reward validity, split and privacy evidence needed to audit any concrete downstream training run.","primary_link":"https://arxiv.org/abs/2508.03680","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/agent-lightning"},{"key":"project","label":["Project","项目主页"],"url":"https://www.microsoft.com/en-us/research/project/agent-lightning/"}],"link_count":4,"sections":9},{"id":"agent-as-a-judge-2025","title":"Agent-as-a-Judge: Evaluate Agents with Agents","year":2025,"venue":"ICML 2025","authors":["Mingchen Zhuge","Changsheng Zhao","Dylan Ashley","Wenyi Wang","Dmitrii Khizbullin","Yunyang Xiong","Zechun Liu","Ernie Chang","Raghuraman Krishnamoorthi","Yuandong Tian","Yangyang Shi","Vikas Chandra","Jürgen Schmidhuber"],"authors_zh":"Mingchen Zhuge, Changsheng Zhao, Dylan Ashley, Wenyi Wang, Dmitrii Khizbullin, Yunyang Xiong, Zechun Liu, Ernie Chang, Raghuraman Krishnamoorthi, Yuandong Tian, Yangyang Shi, Vikas Chandra, Jürgen Schmidhuber","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"agent 评测框架与分层需求数据集论文","best_for_zh":"需要构建、审计或复现 workspace agent 评测与过程证据判决的研究者。","confidence":"medium","one_line":["An agent-based evaluator that turns task completion and evidence gathering into an auditable judgment procedure.","Agent-as-a-Judge 主动读取 workspace 证据，按分层 requirements 评测开发 agent，并以 DevAI 提供可复核的概念验证。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://proceedings.mlr.press/v267/zhuge25a.html","links":[],"link_count":1,"sections":9},{"id":"agent-r1-2025","title":"Agent-R1: Training Powerful LLM Agents with End-to-End Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["Mingyue Cheng","Jie Ouyang","Shuo Yu","Ruiran Yan","Yucong Luo","Zirui Liu","Daoyu Wang","Qi Liu","Enhong Chen"],"authors_zh":"Mingyue Cheng、Jie Ouyang、Shuo Yu、Ruiran Yan、Yucong Luo、Zirui Liu、Daoyu Wang、Qi Liu、Enhong Chen","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","verifier_reward","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_training","multi_hop_question_answering","tool_use","interactive_environments"],"tags":["agent-r1","agentic-rl","toolenv","multi-hop-qa","grpo","rloo","reinforce-plus-plus","step-level-trajectories","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"多轮工具智能体强化学习、轨迹掩码归因与版本漂移审计","best_for_zh":"关注智能体轨迹、工具环境、程序化奖励、token 掩码、检索可复现性和论文到代码版本漂移的研究者与工程人员","confidence":"high","one_line":["Agent-R1 discloses a tool-environment RL framework and a v1 Multi-hop QA reward contract across PPO/GRPO/REINFORCE++/RLOO, but no immutable v1 train split, environment/corpus snapshot, rollout/reward ledger or provenance-complete checkpoint package was verified.","Agent-R1 v1 披露了工具环境 RL 框架，以及覆盖 PPO/GRPO/REINFORCE++/RLOO 的多跳问答奖励契约；未核验到不可变 v1 训练切分、环境/语料快照、rollout/奖励账本或溯源完整的检查点包。"],"why":"It makes the environment transition boundary and token-mask credit assignment explicit for agentic RL, yet shows why an open framework and named benchmark mixture do not establish replayable trajectories, data rights, verifier decisions or reproducible policy lineage.","primary_link":"https://arxiv.org/pdf/2511.14460v1","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AgentR1/Agent-R1"}],"link_count":4,"sections":9},{"id":"agent-rewardbench-multimodal-agents-2025","title":"Agent-RewardBench: Towards a Unified Benchmark for Reward Modeling across Perception, Planning, and Safety in Real-World Multimodal Agents","year":2025,"venue":"ACL 2025","authors":["Tianyi Men","Zhuoran Jin","Pengfei Cao","Yubo Chen","Kang Liu","Jun Zhao"],"authors_zh":"Tianyi Men, Zhuoran Jin, Pengfei Cao, Yubo Chen, Kang Liu, Jun Zhao","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-agents","step-level-reward","safety"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多模态智能体逐步奖励建模评测基准论文","best_for_zh":"需要比较多模态奖励模型在感知、规划和安全智能体步骤中表现的研究者。","confidence":"high","one_line":["Agent-RewardBench evaluates whether multimodal reward models can choose better agent steps across perception, planning, and safety.","约 1,140 条图文 Agent 样本，横跨感知、规划和安全，提供分场景的优劣行为比较与奖励评测。"],"why":"It makes reward quality measurable at the point where an agent’s trajectory can still be corrected.","primary_link":"https://aclanthology.org/2025.acl-long.857/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Quester-one/Agent-RewardBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MultimodalAgent/Agent-RewardBench"}],"link_count":5,"sections":9},{"id":"agent-rlvr-2025","title":"Agent-RLVR: Training Software Engineering Agents via Guidance and Environment Rewards","year":2025,"venue":"arXiv preprint","authors":["Jeff Da","Clinton Wang","Xiang Deng","Yuntao Ma","Nikhil Barhate","Sean Hendryx"],"authors_zh":"Jeff Da、Clinton Wang、Xiang Deng、Yuntao Ma、Nikhil Barhate、Sean Hendryx","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr","agent_training","test_time_compute"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["software_engineering","agentic_reasoning"],"tags":["agent-rlvr","software-engineering-agents","environment-rewards","unit-tests","guided-trajectories","offline-dpo","swe-bench-verified"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["Agent-RLVR collects test-verified software-agent trajectories and teacher-guided reattempts, then reuses same-task positive/negative pairs for SFT, offline DPO, and a test-time patch reward model.","Agent-RLVR 收集经测试验证的软件智能体轨迹与教师引导的重试记录，再把同一任务的正负样本对复用于监督微调、离线 DPO 和推理时的补丁奖励模型。"],"why":"For the rollout/search/test-time trace track, it exposes how repository states, execution outcomes, guidance, reattempts, and pairwise patch feedback can be joined into training and selection records, while leaving the underlying dataset and trajectories unreleased.","primary_link":"https://arxiv.org/abs/2506.11425","links":[{"key":"project","label":["Project","项目主页"],"url":"https://labs.scale.com/papers/agent_rlvr"}],"link_count":4,"sections":9},{"id":"agentauditor-safety-security-evaluation-2025","title":"AgentAuditor: Human-Level Safety and Security Evaluation for LLM Agents","year":2025,"venue":"NeurIPS 2025","authors":["Hanjun Luo","Shenyu Dai","Chiming Ni","Xinfeng Li","Guibin Zhang","Kun Wang","Tongliang Liu","Hanan Salam"],"authors_zh":"Hanjun Luo、Shenyu Dai、Chiming Ni、Xinfeng Li、Guibin Zhang、Kun Wang、Tongliang Liu、Hanan Salam","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["ASSEBench pairs agent trajectories with strict and lenient safety/security judgments, while AgentAuditor retrieves prior cases to strengthen automated auditing.","ASSEBench 为智能体安全与安全防护轨迹提供双标准专家判定，AgentAuditor 以检索式经验记忆增强自动审计。"],"why":"ASSEBench pairs agent trajectories with strict and lenient safety/security judgments, while AgentAuditor retrieves prior cases to strengthen automated auditing.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/3dc85735f6e2fcf093e67b134fa00d21-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Astarojth/AgentAuditor-ASSEBench"}],"link_count":2,"sections":9},{"id":"agent-change-bench-2025","title":"AgentChangeBench: A Multi-Dimensional Evaluation Framework for Goal-Shift Robustness in Conversational AI","year":2025,"venue":"NeurIPS 2025 Workshop on Multi-Turn Interactions in Large Language Models (MTI-LLM)","authors":["Manik Rana","Calissa Man","Anotida Expected Msiiwa","Jeffrey Paine","Kevin Zhu","Sunishchal Dev","Vasu Sharma","Ahan M R"],"authors_zh":"Manik Rana, Calissa Man, Anotida Expected Msiiwa, Jeffrey Paine, Kevin Zhu, Sunishchal Dev, Vasu Sharma, Ahan M R","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","conversational_agents","customer_service","banking","airline","retail"],"tags":["environment-agent-trajectory-data","agent-benchmark","conversational-agent","tool-use","goal-shift","user-simulator","llm-as-judge","trajectory-evaluation","replay-risk","version-drift"],"status":"partial","priority":"可读","paper_type_zh":"目标变更对话智能体环境评测基准","best_for_zh":"研究 tool-use agent、用户模拟器、goal-shift 评测与混合反馈审计的读者","confidence":"medium","one_line":["AgentChangeBench evaluates 315 banking, airline, and retail goal-shift tasks through simulated tool-use episodes and mixed environment/LLM metrics, but its 2,835-sequence mapping and supplementary release remain insufficiently pinned for replay or training reuse.","AgentChangeBench 用 315 个银行、航空与零售 goal-shift 任务及混合环境/LLM 指标评测对话 agent，但 2,835 条 sequence 的映射和补充发布不足以支持 replay 或训练复用。"],"why":"It separates final task progress, tool validity, repeated actions, and post-shift adaptation instead of collapsing an agent episode into pass/fail, while exposing the audit burden created by LLM judges, closed-model simulators, inherited public tasks, and unversioned environment artifacts.","primary_link":"https://arxiv.org/abs/2510.18170","links":[{"key":"data","label":["Data","数据"],"url":"https://openreview.net/attachment?id=ZCi58UP9uR&name=supplementary_material"}],"link_count":6,"sections":9},{"id":"agentgym-2024","title":"AgentGym: Evaluating and Training Large Language Model-based Agents across Diverse Environments","year":2025,"venue":"ACL 2025 Long Papers","authors":["Zhiheng Xi","Yiwen Ding","Wenxiang Chen","Boyang Hong","Honglin Guo","Junzhe Wang","Xin Guo","Dingwen Yang","Chenyang Liao","Wei He","Songyang Gao","Lu Chen","Rui Zheng","Yicheng Zou","Tao Gui","Qi Zhang","Xipeng Qiu","Xuanjing Huang","Zuxuan Wu","Yu-Gang Jiang"],"authors_zh":"Zhiheng Xi、Yiwen Ding、Wenxiang Chen、Boyang Hong、Honglin Guo、Junzhe Wang、Xin Guo 等（Fudan University / Fudan NLP Lab / Fudan Vision and Learning Lab / Peng Cheng Laboratory）","tracks":["benchmarks_evaluation_surfaces","process_trace_supervision_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["agent_training","evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["tool-use","agent-environments","interactive-evaluation"],"tags":["tool-use","agent-environment","trajectory-data"],"status":"verified","priority":"可读","paper_type_zh":"Agent 环境 / 工具使用数据与评测","best_for_zh":"研究多环境 agent evaluation、trajectory 数据、环境反馈契约、behavioral cloning 和 self-evolution 的读者","confidence":"medium","one_line":["AgentGym contributes a 14-environment agent platform with AgentEval, AgentTraj/AgentTraj-L, and AgentEvol as linked evaluation, trajectory, and training artifacts.","AgentGym 将 14 个交互环境、AgentEval 评测任务、AgentTraj/AgentTraj-L 轨迹和 AgentEvol 训练方法组织成一个可审计的 agent 平台。"],"why":"It exposes action schema, observation state, environment feedback, reward/success predicates, and train/eval split risks for agent-data curation.","primary_link":"https://aclanthology.org/2025.acl-long.1355/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WooooDyy/AgentGym"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AgentGym/AgentTraj-L"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/AgentGym/AgentEval"},{"key":"project","label":["Project","项目主页"],"url":"https://agentgym.github.io"}],"link_count":7,"sections":9},{"id":"agentic-reward-modeling-2025","title":"Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems","year":2025,"venue":"ACL 2025","authors":["Hao Peng","Yunjia Qi","Xiaozhi Wang","Zijun Yao","Bin Xu","Lei Hou","Juanzi Li"],"authors_zh":"Hao Peng, Yunjia Qi, Xiaozhi Wang, Zijun Yao, Bin Xu, Lei Hou, Juanzi Li","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","reward-model","public-artifact"],"status":"verified","priority":"必读","paper_type_zh":"可验证信号增强的奖励建模与偏好优化论文","best_for_zh":"需要将人类偏好与可执行或可核验信号结合为可靠奖励的研究者。","confidence":"high","one_line":["A reward-agent framework that combines a preference reward model with factuality and instruction-following verifiers, and releases its implementation.","以偏好奖励结合事实性与指令遵循验证器，构造可审计的复合奖励代理。"],"why":"It provides an auditable public surface for measuring or mitigating reward and judge reliability failures.","primary_link":"https://aclanthology.org/2025.acl-long.775/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/Agentic-Reward-Modeling"}],"link_count":3,"sections":9},{"id":"agentif-agentic-instruction-following-2025","title":"AGENTIF: Benchmarking Instruction Following of Large Language Models in Agentic Scenarios","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Yunjia Qi","Hao Peng","Xiaozhi Wang","Amy Xin","Youfeng Liu","Bin Xu","Lei Hou","Juanzi Li"],"authors_zh":"Yunjia Qi、Hao Peng、Xiaozhi Wang、Amy Xin、Youfeng Liu、Bin Xu、Lei Hou、Juanzi Li","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"必读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["AGENTIF turns long agent instructions from real applications into atomic constraints with auditable verifiers.","从真实 Agent 应用提取长指令与原子约束，逐项衡量模型能否完整遵循。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/51bb3a8a33610a25aae074bfc51b1b1f-Paper-Datasets_and_Benchmarks_Track.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/AgentIF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THU-KEG/AgentIF"}],"link_count":4,"sections":9},{"id":"agent-reward-bench-2025","title":"AgentRewardBench: Evaluating Automatic Evaluations of Web Agent Trajectories","year":2025,"venue":"Conference on Language Modeling (COLM) 2025","authors":["Xing Han Lù","Amirhossein Kazemnejad","Nicholas Meade","Arkil Patel","Dongchan Shin","Alejandra Zambrano","Karolina Stańczak","Peter Shaw","Christopher J. Pal","Siva Reddy"],"authors_zh":"Xing Han Lù、Amirhossein Kazemnejad、Nicholas Meade、Arkil Patel、Dongchan Shin、Alejandra Zambrano、Karolina Stańczak、Peter Shaw、Christopher J. Pal、Siva Reddy","tracks":["environment_agent_trajectory_data","preference_reward_feedback_data"],"source_role":["benchmark","data_release","verifier_reward","audit_failure"],"verification_contract":["mixed","judgment_required","environmental"],"supervision_granularity":["full_episode","scalar_reward","trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["web_agents","browser_navigation","trajectory_evaluation","llm_judges","enterprise_web_workflows","real_world_web"],"tags":["environment-agent-trajectory-data","web-agent","browser-agent","trajectory-evaluation","llm-as-judge","expert-annotation","rule-based-evaluator","reward-audit","failed-trajectories","replay-risk","license-risk"],"status":"partial","priority":"必读","paper_type_zh":"网页智能体轨迹评估基准与反馈审计数据发布","best_for_zh":"研究网页智能体评测、轨迹反馈契约、LLM-as-a-judge 与环境奖励审计的读者","confidence":"high","one_line":["AgentRewardBench releases 1,302 web-agent episodes with expert success, side-effect, optimality, and repetition labels plus 15 automatic-evaluator outputs per episode, but it supports offline audit rather than licensed deterministic replay.","AgentRewardBench 发布 1,302 条网页智能体完整轨迹、专家标签与每条轨迹 15 组自动评估输出，用于审计 LLM judge 和环境规则，但不提供已获许可的确定性环境重放。"],"why":"It lets researchers measure false-positive and false-negative behavior in both LLM judges and task-specific rules before those signals are used to select agent trajectories or report success, while preserving failed examples and exposing release/replay risks.","primary_link":"https://arxiv.org/abs/2504.08942","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/McGill-NLP/agent-reward-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/McGill-NLP/agent-reward-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://agent-reward-bench.github.io/"}],"link_count":13,"sections":9},{"id":"agentrm-2025","title":"AgentRM: Enhancing Agent Generalization with Reward Modeling","year":2025,"venue":"ACL 2025","authors":["Yu Xia","Jingru Fan","Weize Chen","Siyu Yan","Xin Cong","Zhong Zhang","Yaxi Lu","Yankai Lin","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Yu Xia, Jingru Fan, Weize Chen, Siyu Yan, Xin Cong, Zhong Zhang, Yaxi Lu, Yankai Lin, Zhiyuan Liu, Maosong Sun","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","reward-model","public-artifact"],"status":"verified","priority":"必读","paper_type_zh":"智能体过程奖励与测试时搜索评测论文","best_for_zh":"需要审计奖励信号迁移性和智能体搜索可靠性的研究者。","confidence":"high","one_line":["An 8B reward model with released code and data for testing whether reward-guided search generalizes across unseen agent tasks.","用搜索状态价值训练通用过程奖励，以测试时搜索提升智能体跨任务表现。"],"why":"It provides an auditable public surface for measuring or mitigating reward and judge reliability failures.","primary_link":"https://aclanthology.org/2025.acl-long.945/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thunlp/AgentRM"}],"link_count":2,"sections":9},{"id":"agenttrek-2025","title":"AgentTrek: Agent Trajectory Synthesis via Guiding Replay with Web Tutorials","year":2025,"venue":"ICLR 2025 Spotlight","authors":["Yiheng Xu","Dunjie Lu","Zhennan Shen","Junli Wang","Zekun Wang","Yuchen Mao","Caiming Xiong","Tao Yu"],"authors_zh":"Yiheng Xu、Dunjie Lu、Zhennan Shen、Junli Wang、Zekun Wang、Yuchen Mao、Caiming Xiong、Tao Yu","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release","agent_environment"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["web_agents","gui_agents","browser_automation","multimodal_reasoning"],"tags":["agenttrek","web-agents","gui-agents","web-tutorials","redpajama","guided-replay","browsergym","playwright","vlm-judge","synthetic-trajectories","sft","environment-drift","count-reconciliation","copyright-risk","privacy-risk","release-completeness"],"status":"partial","priority":"必读","paper_type_zh":"以网页教程驱动的智能体轨迹构造配方、数据发布与实时浏览器环境报告","best_for_zh":"研究 GUI/web agent SFT、tutorial-to-trajectory synthesis、VLM judge filtering、live-site replay，或审计 agent data release 完整性的读者","confidence":"high","one_line":["AgentTrek mines and structures RedPajama tutorials, executes them through GPT-4o-guided BrowserGym replay on 127 sites, and filters trajectories with a GPT-4o evaluator for GUI-agent SFT.","AgentTrek 将 RedPajama 教程经筛选、结构化、live-browser guided replay 与 GPT-4o judgment 转为 GUI-agent SFT 数据；论文主张 23,430 个教程产生 10,398 条成功 trajectory，但当前 HF 只发布 52,594 条无 trajectory 映射的 text turn，未提供完整 multimodal/native traces、失败数据、license 或环境 manifest。"],"why":"It demonstrates a scalable bridge from procedural web text to grounded agent behavior while exposing why judge semantics, live-site drift, rejected data, source rights, privacy, unit mapping, and release manifests determine whether an open trajectory dataset is auditable.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/c681fb2bf1d785fbc766f3ea14758aab-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xlang-ai/AgentTrek"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xlangai/AgentTrek"},{"key":"project","label":["Project","项目主页"],"url":"https://agenttrek.github.io/"}],"link_count":14,"sections":9},{"id":"agenttts-compute-allocation-2025","title":"AgentTTS: Large Language Model Agent for Test-time Compute-optimal Scaling Strategy in Complex Tasks","year":2025,"venue":"NeurIPS 2025","authors":["Fali Wang","Hui Liu","Zhenwei Dai","Jingying Zeng","Zhiwei Zhang","Zongyu Wu","Chen Luo","Zhen Li","Xianfeng Tang","Qi He","Suhang Wang"],"authors_zh":"Fali Wang、Hui Liu、Zhenwei Dai、Jingying Zeng、Zhiwei Zhang、Zongyu Wu、Chen Luo、Zhen Li、Xianfeng Tang、Qi He、Suhang Wang（机构：The Pennsylvania State University、Amazon）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["general-reasoning","software-engineering"],"tags":["test-time-compute","budget-allocation","multi-stage-tasks","agent","repeated-sampling","fusion"],"status":"verified","priority":"必读","paper_type_zh":"多阶段测试时计算分配研究（NeurIPS 2025）","best_for_zh":"研究多步骤语言模型工作流中模型规模、采样次数与总推理预算分配的读者。","confidence":"high","one_line":["AgentTTS uses an LLM agent and execution feedback to find compute-optimal model and sampling allocations across dependent stages of a complex task.","AgentTTS 通过语言模型智能体和执行反馈，在多阶段复杂任务中搜索模型选择与重复采样预算的计算最优分配。"],"why":"It turns test-time scaling from one-model sampling into a joint decision over stage order, model size, and sampling budgets.","primary_link":"https://arxiv.org/abs/2508.00890","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/FairyFali/AgentTTS"},{"key":"data","label":["Data","数据"],"url":"https://github.com/FairyFali/AgentTTS/tree/main/data"}],"link_count":5,"sections":9},{"id":"aimo2-openmathreasoning-2025","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","year":2025,"venue":"arXiv report of the AIMO-2 winning submission; retitled poster at the 2nd AI for Math Workshop @ ICML 2025","authors":["Ivan Moshkov","Darragh Hanley","Ivan Sorokin","Shubham Toshniwal","Christof Henkel","Benedikt Schifferer","Wei Du","Igor Gitman"],"authors_zh":"Ivan Moshkov、Darragh Hanley、Ivan Sorokin、Shubham Toshniwal、Christof Henkel、Benedikt Schifferer、Wei Du、Igor Gitman","tracks":["data_construction_open_release_recipes","programmatically_verifiable_outcome_data","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed","judgment_required","programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics","olympiad_mathematics","tool_integrated_reasoning"],"tags":["openmathreasoning","aimo-2","mathematical-reasoning","deepseek-r1-distillation","tool-integrated-reasoning","genselect","answer-level-verification","open-data-recipe"],"status":"partial","priority":"必读","paper_type_zh":"数学推理数据构造、工具交互轨迹、生成式选择与开放发布","best_for_zh":"构建数学 SFT/TIR/GenSelect 数据、审计 answer-level filtering，或研究论坛抓取、发布计数和版本漂移风险的读者","confidence":"high","one_line":["OpenMathReasoning packages 5.68M rows across CoT, Python-interleaved TIR, GenSelect, and problem-only modes for answer-level mathematical-reasoning SFT and selection, but the official release corrects the paper's 540K problem figure to 306K unique solution-bearing problems.","OpenMathReasoning 发布 5,678,317 条 CoT、Python-TIR、GenSelect 与 problem-only 记录；官方数据卡把论文早期 540K 题目口径更正为 306K 个有 solution 的唯一题目，另含 193,170 条仅题目记录，且仍缺原始 AoPS 抓取与逐条 verifier lineage。"],"why":"It exposes a rare end-to-end recipe from forum problems through teacher rollouts, answer judging, tool-use filtering, generative selection, and SFT. Its release correction and missing raw-scrape, rejection, verifier, and rights manifests also show why dataset lineage and acceptance semantics must be audited rather than inferred from aggregate scale.","primary_link":"https://arxiv.org/abs/2504.16891","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Skills/tree/main/recipes/openmathreasoning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenMathReasoning"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/nvidia/openmathreasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://nvidia-nemo.github.io/Skills/releases/openmathreasoning/"}],"link_count":8,"sections":9},{"id":"ainsteinbench-scientific-repository-agents-2025","title":"AInsteinBench: Benchmarking Coding Agents on Scientific Repositories","year":2025,"venue":"arXiv","authors":["Titouan Duston","Shuo Xin","Yang Sun","Daoguang Zan","Aoyan Li","Shulin Xin","Kai Shen","Yixiao Chen","Qiming Sun","Ge Zhang","Jiashuo Liu","Huan Zhou","Jingkai Liu","Zhichen Pu","Yuanheng Wang","Bo-Xuan Ge","Xin Tong","Fei Ye","Zhi-Chao Zhao","Wen-Biao Han","Zhoujian Cao","Yueran Zhao","Weiluo Ren","Qingshen Long","Yuxiao Liu","Anni Huang","Yidi Du","Yuanyuan Rong","Jiahao Peng"],"authors_zh":"Titouan Duston, Shuo Xin, Yang Sun, Daoguang Zan, Aoyan Li, Shulin Xin, Kai Shen, Yixiao Chen, Qiming Sun, Ge Zhang, Jiashuo Liu, Huan Zhou, Jingkai Liu, Zhichen Pu, Yuanheng Wang, Bo-Xuan Ge, Xin Tong, Fei Ye, Zhi-Chao Zhao, Wen-Biao Han, Zhoujian Cao, Yueran Zhao, Weiluo Ren, Qingshen Long, Yuxiao Liu, Anni Huang, Yidi Du, Yuanyuan Rong, Jiahao Peng","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","agent_training"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["scientific-computing","software-engineering","agent-evaluation"],"tags":["scientific-computing","software-engineering","agent-evaluation","unit-tests","2025"],"status":"verified","priority":"可读","paper_type_zh":"科学计算软件仓库的可执行智能体评测基准","best_for_zh":"需要在真实科研软件环境中检验代码智能体正确性的研究者。","confidence":"high","one_line":["AInsteinBench evaluates coding agents on executable scientific-repository tasks derived from maintainer pull requests and checked by task-specific tests.","AInsteinBench 从科学软件仓库维护者的 PR 构建可执行任务，并以任务测试检验智能体的代码修改。"],"why":"It evaluates edits in runnable scientific environments rather than abstract programming questions.","primary_link":"https://arxiv.org/abs/2512.21373","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ByteDance-Seed/AInsteinBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ByteDance-Seed/AInsteinBench"}],"link_count":4,"sections":9},{"id":"physics-aware-rs-materials-2025","title":"Aligning Reasoning LLMs for Materials Discovery with Physics-aware Rejection Sampling","year":2025,"venue":"NeurIPS 2025 Workshop on AI for Science: The Reach and Limits of AI for Scientific Discovery (poster)","authors":["Lee Hyun","Sohee Yoon","Jinwoo Park","Sue In Chae","Seongeon Park","Jooyeon Ahn","Yebin Jung","Youjung Chung","Hogeun Chang","Sujin Park","Myeonginn Kang","Jina Kim","Ho-Gyeong Kim","Myeonghun Jeong"],"authors_zh":"Lee Hyun、Sohee Yoon、Jinwoo Park、Sue In Chae、Seongeon Park、Jooyeon Ahn、Yebin Jung、Youjung Chung、Hogeun Chang、Sujin Park、Myeonginn Kang、Jina Kim、Ho-Gyeong Kim、Myeonghun Jeong","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["materials_science","scientific_reasoning"],"tags":["materials-discovery","qd-led","rejection-sampling","physics-aware","teacher-traces","sft-distillation","adaptive-halting"],"status":"partial","priority":"可读","paper_type_zh":"材料配方推理的物理约束拒绝采样与 SFT 蒸馏研究","best_for_zh":"需要审计科学领域 teacher traces、物理 verifier、拒绝采样与早停归因的研究者","confidence":"high","one_line":["PaRS samples teacher QD-LED reasoning traces and keeps an early candidate only if range, wet-lab-error, and PLQY-envelope gates pass before distilling Qwen3-32B.","PaRS 为 QD-LED 配方采样教师推理轨迹，仅保留同时通过范围、湿实验误差和 PLQY 上包络门的早期候选，再蒸馏 Qwen3-32B；但轨迹与门控审计记录未发布。"],"why":"It demonstrates a physics-aware trace-selection contract beyond final-answer correctness, but unavailable recipes, traces, gate logs, thresholds, and wet-lab provenance block independent reuse or audit.","primary_link":"https://arxiv.org/abs/2509.00768","links":[],"link_count":4,"sections":9},{"id":"amazon-nova-2-sonic-service-card-2025","title":"Amazon Nova 2 Sonic - AWS AI Service Cards","year":2025,"venue":"AWS AI Service Card","authors":["Amazon Web Services"],"authors_zh":"Amazon Web Services","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","multimodal","agentic_tool_use","safety"],"tags":["amazon","aws","nova-2-sonic","service-card","frontier-report","data-disclosure-ledger","sft","rlhf","safety-evaluation","multimodal","speech"],"status":"partial","priority":"必读","paper_type_zh":"云服务模型卡与数据披露台账","best_for_zh":"审查云端语音模型的数据、对齐、guardrails 和审计边界的读者","confidence":"high","one_line":["AWS's Nova 2 Sonic card discloses high-level multimodal sources, SFT/RL/RLHF, runtime filters, and evaluation layers but not records, rights, rewards, or reproducibility artifacts.","AWS 的 Nova 2 Sonic 服务卡披露了高层多模态来源、SFT/RL/RLHF、运行时过滤与评测层，但没有公开记录、权利、奖励或可复现制品。"],"why":"A speech-model disclosure baseline separating high-level safety and alignment claims from unavailable audit evidence.","primary_link":"https://docs.aws.amazon.com/ai/responsible-ai/nova-2-sonic/overview.html","links":[],"link_count":2,"sections":9},{"id":"aao-semantic-disambiguation-dpo-2025","title":"Ambiguity Awareness Optimization: Towards Semantic Disambiguation for Direct Preference Optimization","year":2025,"venue":"EMNLP 2025","authors":["Jian Li","Shenglin Yin","Yujia Zhang","Alan Zhao","Xi Chen","Xiaohui Zhou","Pengfei Xu"],"authors_zh":"Jian Li、Shenglin Yin、Yujia Zhang、Alan Zhao、Xi Chen、Xiaohui Zhou、Pengfei Xu（腾讯 OVB 人工智能技术中心、北京大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","token-weighting","semantic-similarity","dpo"],"status":"verified","priority":"必读","paper_type_zh":"词元级偏好数据加权与优化目标研究","best_for_zh":"适合研究当优选与劣选文本大量表达相同内容时，偏好回答对是否应被均匀消费的读者。","confidence":"high","one_line":["AAO reweights tokens in DPO preference pairs using their semantic similarity across chosen and rejected responses, reducing ambiguous supervision and emphasizing discriminative content.","AAO 根据优选与劣选回答之间的语义相似度为 DPO 回答对中的词元赋权，压制含混监督并强调区分性内容。"],"why":"It makes semantic redundancy within a preference pair an auditable token-level training signal instead of letting identical content receive opposing DPO gradients.","primary_link":"https://aclanthology.org/2025.emnlp-main.460/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InsLin/AAO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuggingFaceH4/ultrafeedback_binarized"}],"link_count":4,"sections":9},{"id":"amex-mobile-gui-agents-2025","title":"AMEX: Android Multi-annotation Expo Dataset for Mobile GUI Agents","year":2025,"venue":"Findings of ACL 2025 / arXiv","authors":["Yuxiang Chai","Siyuan Huang","Yazhe Niu","Han Xiao","Liang Liu","Guozhi Wang","Dingyu Zhang","Shuai Ren","Hongsheng Li"],"authors_zh":"Yuxiang Chai 等（MMLab, The Chinese University of Hong Kong）","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["mobile-agents","gui-annotation","android"],"tags":["agent_environment","trajectory_data","mobile-agents","gui-annotation","android"],"status":"verified","priority":"暂缓","paper_type_zh":"Findings of ACL 2025 / arXiv 的 mobile GUI annotation dataset","best_for_zh":"关注移动端、GUI、OS、跨环境智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["AMEX provides Android GUI screenshots with multi-annotation labels and action chains.","AMEX 提供 Android GUI 截图、多层标注和动作链。"],"why":"it gives multi-layer GUI annotations and action chains for mobile agents","primary_link":"https://arxiv.org/abs/2407.17490","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YuxiangChai/AMEX-codebase"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Yuxiang007/AMEX"},{"key":"project","label":["Project","项目主页"],"url":"https://www.yxchai.com/AMEX"}],"link_count":5,"sections":9},{"id":"epicprm-epic50k-process-supervision-data-2025","title":"An Efficient and Precise Training Data Construction Framework for Process-supervised Reward Model in Mathematical Reasoning","year":2025,"venue":"arXiv","authors":["Wei Sun","Qianlong Du","Fuwei Cui","Jiajun Zhang"],"authors_zh":"Li et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["An Efficient and Precise Training Data Construction Framework for Process-supervised Reward Model in Mathematical Reasoning exposes process or trace supervision data.","提出 EpicPRM 数据构建框架，并公开 Epic50K：通过贡献度标注和自适应二分首错定位生成 5 万条高质量过程监督样本。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2503.02382","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xiaolizh1/EpicPRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SunW7777/EpicPRM"}],"link_count":3,"sections":9},{"id":"an-empirical-study-llm-judge-2025","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation","year":2025,"venue":"Findings of ACL 2025","authors":["Hui Huang","Xingyuan Bu","Hongli Zhou","Yingqi Qu","Jing Liu","Muyun Yang","Bing Xu","Tiejun Zhao"],"authors_zh":"Hui Huang, Xingyuan Bu, Hongli Zhou, Yingqi Qu, Jing Liu, Muyun Yang, Bing Xu, Tiejun Zhao","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"微调判别模型可靠性审计","best_for_zh":"需要选择、微调或审计 LLM-as-a-Judge 的评测、对齐与安全研究者。","confidence":"high","one_line":["Audits whether fine-tuned LLM judges generalize beyond their training evaluation format, and finds they often behave as task-specific classifiers.","审计微调判别模型能否跨越原训练评测格式；结果显示其常退化为任务特定分类器。"],"why":"It shows why strong in-domain judge scores can conceal failures in transfer, fairness, and prompt adaptability.","primary_link":"https://aclanthology.org/2025.findings-acl.306/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HuihuiChyan/UnlimitedJudge"}],"link_count":2,"sections":9},{"id":"online-mind2web-2025","title":"An Illusion of Progress? Assessing the Current State of Web Agents","year":2025,"venue":"COLM 2025","authors":["Tianci Xue","Weijian Qi","Tianneng Shi","Chan Hee Song","Boyu Gou","Dawn Song","Huan Sun","Yu Su"],"authors_zh":"Tianci Xue, Weijian Qi, Tianneng Shi, Chan Hee Song, Boyu Gou, Dawn Song, Huan Sun, Yu Su","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["agent_environment","benchmark","data_release","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","web_agents","live_web","gui_control"],"tags":["online-mind2web","live-web","web-agent","agent-trajectory","webjudge","llm-as-judge","reward-model","screenshot-selection","task-versioning","replay-risk","contamination-risk"],"status":"partial","priority":"必读","paper_type_zh":"在线网页智能体基准、轨迹评测与奖励验证器","best_for_zh":"研究在线网页智能体、视觉轨迹评测、LLM judge、奖励模型、任务漂移与可回放性的读者","confidence":"high","one_line":["Online-Mind2Web releases 300 mutable live-site tasks, evaluation code and task-level labels plus WebJudge-7B, but not a complete immutable corpus of all paper trajectories or the external website states needed for exact replay.","Online-Mind2Web 发布 300 个会随真实网站变化的任务、评测代码、任务级标签与 WebJudge-7B，但未形成包含六类论文智能体全部截图/动作及网站快照的不可变轨迹语料。"],"why":"It couples realistic browser episodes with human and judge feedback while exposing how screenshot selection, task replacement, web drift, execution failures, judge choice, and incomplete trajectory retention affect evaluation and reward-model reuse.","primary_link":"https://arxiv.org/abs/2504.01382","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OSU-NLP-Group/Online-Mind2Web"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/osunlp/Online-Mind2Web"}],"link_count":16,"sections":9},{"id":"llm-judge-uncertainty-2025","title":"Analyzing Uncertainty of LLM-as-a-Judge: Interval Evaluations with Conformal Prediction","year":2025,"venue":"EMNLP 2025","authors":["Huanxin Sheng","Xinyi Liu","Hangfeng He","Jieyu Zhao","Jian Kang"],"authors_zh":"Huanxin Sheng, Xinyi Liu, Hangfeng He, Jieyu Zhao, Jian Kang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["judge-reliability","uncertainty-quantification","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"EMNLP 2025 的 LLM judge 不确定性量化与评测可靠性研究","best_for_zh":"需要为自动评测器报告校准不确定性、审计评分可靠性的研究者。","confidence":"high","one_line":["judge uncertainty evaluation data and conformal prediction code","该工作以保形预测为评分型 LLM judge 输出覆盖率可检验的区间与中点分数。"],"why":"It makes uncertainty in LLM-as-a-judge evaluation measurable and auditable.","primary_link":"https://aclanthology.org/2025.emnlp-main.569/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/BruceSheng1202/Analyzing-Uncertainty-of-LLM-as-a-Judge"}],"link_count":2,"sections":9},{"id":"apo-clair-underspecification-2025","title":"Anchored Preference Optimization and Contrastive Revisions: Addressing Underspecification in Alignment","year":2025,"venue":"Transactions of the Association for Computational Linguistics","authors":["Karel D’Oosterlinck","Winnie Xu","Chris Develder","Thomas Demeester","Amanpreet Singh","Christopher Potts","Douwe Kiela","Shikib Mehri"],"authors_zh":"Karel D’Oosterlinck、Winnie Xu、Chris Develder、Thomas Demeester、Amanpreet Singh、Christopher Potts、Douwe Kiela、Shikib Mehri","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["trace_writing","reward_verifier_layer"],"domains":["alignment","instruction-following"],"tags":["alignment","preference-optimization","contrastive-revision","clair"],"status":"verified","priority":"可读","paper_type_zh":"偏好优化与对比式修订数据论文","best_for_zh":"研究偏好优化、指令对齐或反馈数据构造的读者。","confidence":"high","one_line":["APO and CLAIR use contrastive revisions to make preference optimization more robust to underspecified alignment feedback.","APO 与 CLAIR 利用对比式修订数据缓解偏好反馈对齐中的欠确定性。"],"why":"The dataset represents an actionable revision-based alternative to treating a single preference label as fully specified.","primary_link":"https://aclanthology.org/2025.tacl-1.22/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ContextualAI/CLAIR_and_APO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ContextualAI/ultrafeedback_clair_32k"}],"link_count":4,"sections":9},{"id":"android-control-curated-2025","title":"AndroidControl-Curated: Revealing the True Potential of GUI Agents through Benchmark Purification","year":2025,"venue":"arXiv preprint","authors":["Ho Fai Leung","Xiaoyan Xi","Fei Zuo"],"authors_zh":"Ho Fai Leung, Xiaoyan Xi, Fei Zuo","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release","construction_recipe","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level"],"training_use":["agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","mobile_ui_control","gui_grounding"],"tags":["environment-agent-trajectory-data","benchmarks-evaluation-surfaces","android","gui-agent","benchmark-purification","data-quality","grounding-evaluation"],"status":"partial","priority":"可读","paper_type_zh":"Android GUI 基准净化、数据发布与模型报告","best_for_zh":"研究移动 GUI agent 轨迹数据、grounding verifier、benchmark purification 与训练评测泄漏的读者","confidence":"medium","one_line":["AndroidControl-Curated releases five static Android step-data views and screenshot archives, couples box-aligned scoring with LLM-human task and label correction, and uses 2,400 selected samples for GRPO agent training.","AndroidControl-Curated 将 AndroidControl 净化为五种静态 step 数据视图，以 box 对齐评分和 LLM-人工任务/标签修订支撑评测与 Magma-R1 训练，但训练重叠、许可证和可回放环境仍未披露。"],"why":"It demonstrates that the evaluation contract and label quality can move GUI-agent success rates by double digits, while its released records and scorer make those effects auditable enough to study—provided users preserve the unresolved licensing, replay, split, and checkpoint boundaries.","primary_link":"https://arxiv.org/abs/2510.18488","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/batechworks/AndroidControl_Curated"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/batwBMW/AndroidControl_Curated"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/batwBMW/Magma-R1-4B-AndroidControl"}],"link_count":5,"sections":9},{"id":"androidlab-2025","title":"AndroidLab: Training and Systematic Benchmarking of Android Autonomous Agents","year":2025,"venue":"ACL 2025","authors":["Yifan Xu","Xiao Liu","Xueqiao Sun","Siyi Cheng","Hao Yu","Hanyu Lai","Shudan Zhang","Dan Zhang","Jie Tang","Yuxiao Dong"],"authors_zh":"Yifan Xu, Xiao Liu, Xueqiao Sun, Siyi Cheng, Hao Yu, Hanyu Lai, Shudan Zhang, Dan Zhang, Jie Tang, Yuxiao Dong","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","mobile_ui","android","ui_control","multimodal_agents"],"tags":["androidlab","mobile-ui","android","agent-trajectories","human-demonstrations","instruction-tuning","xml-observations","set-of-marks","task-specific-verifiers","llm-judge","replay-partial","release-incomplete","data-license-unknown"],"status":"partial","priority":"可读","paper_type_zh":"Android agent 环境、轨迹数据发布与系统评测","best_for_zh":"研究移动 GUI agent 轨迹构造、SFT、环境 verifier、可回放评测与发布审计的读者","confidence":"medium","one_line":["AndroidLab releases a 726-trajectory Android SFT subset and a 138-task AVD benchmark with UI/device-state and LLM-judged terminal feedback, while splits, failed traces, data rights, and immutable replay remain unresolved.","AndroidLab 发布 726 条 Android SFT 轨迹及含混合终态反馈的 138 任务 AVD 基准，但数据划分、失败轨迹、数据权利与不可变回放条件仍未解决。"],"why":"It exposes how a mobile agent's training record connects task, XML or screenshot state, action, history, and completion feedback; it also shows why app/version pinning, alternate-valid-path verification, failure retention, train/eval overlap, and dataset licensing determine whether such trajectories can be safely reused.","primary_link":"https://aclanthology.org/2025.acl-long.107/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/Android-Lab"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/file/d/1s0b74VEOww9n1kMocd6RJivwaUCymEs4/view?usp=drive_link"}],"link_count":9,"sections":9},{"id":"answer-convergence-early-stopping-2025","title":"Answer Convergence as a Signal for Early Stopping in Reasoning","year":2025,"venue":"EMNLP 2025","authors":["Xin Liu","Lu Wang"],"authors_zh":"Xin Liu、Lu Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["scaling_study","construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["step_level"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["mathematics","question_answering","science"],"tags":["answer-convergence","early-stopping","self-consistency"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["Stops CoT when partial-chain answers stabilize, optionally using logit adjustment or a learned stop predictor.","当部分思维链的答案趋于稳定时提前终止推理，可选配 logit 调整或学习式停止预测器。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"It separates answer stability from verified correctness when selecting trace prefixes.","primary_link":"https://aclanthology.org/2025.emnlp-main.904/","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/spaces/launch/reasoning_earlystop"}],"link_count":5,"sections":9},{"id":"antileakbench-2025","title":"AntiLeakBench: Preventing Data Contamination by Automatically Generating Benchmark Data","year":2025,"venue":"ACL 2025","authors":["Xiaobao Wu","Liangming Pan","Yuxi Xie","Ruiwen Zhou","Shuai Zhao","Yubo Ma","Mingzhe Du","Rui Mao","Anh Tuan Luu","William Yang Wang"],"authors_zh":"Xiaobao Wu, Liangming Pan, Yuxi Xie, Ruiwen Zhou, Shuai Zhao, Yubo Ma, Mingzhe Du, Rui Mao, Anh Tuan Luu, William Yang Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器失效或奖励投机审计论文","best_for_zh":"需要审计基准污染、评估偏差或奖励投机风险的研究者。","confidence":"high","one_line":["automatically refreshed contamination-resistant benchmark data","从截止时间后的事实变化自动构建并更新带来源证据的抗污染问答基准。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://aclanthology.org/2025.acl-long.901/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bobxwu/AntiLeakBench"}],"link_count":2,"sections":9},{"id":"apigen-mt-2025","title":"APIGen-MT: Agentic Pipeline for Multi-Turn Data Generation via Simulated Agent-Human Interplay","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Akshara Prabhakar","Zuxin Liu","Ming Zhu","Jianguo Zhang","Tulika Manoj Awalgaonkar","Shiyu Wang","Zhiwei Liu","Haolin Chen","Thai Hoang","Juan Carlos Niebles","Shelby Heinecke","Weiran Yao","Huan Wang","Silvio Savarese","Caiming Xiong"],"authors_zh":"Akshara Prabhakar、Zuxin Liu、Ming Zhu、Jianguo Zhang、Tulika Manoj Awalgaonkar、Shiyu Wang、Zhiwei Liu、Haolin Chen、Thai Hoang、Juan Carlos Niebles、Shelby Heinecke、Weiran Yao、Huan Wang、Silvio Savarese、Caiming Xiong","tracks":["environment_agent_trajectory_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer"],"domains":["tool_use","function_calling","multi_turn_agents","environment_interaction","customer_service","retail","airline"],"tags":["apigen-mt","environment-agent-trajectory-data","multi-turn","tool-use","function-calling","synthetic-data","tau-bench","simulated-user","success-only-filtering","mixed-verification","partial-release"],"status":"partial","priority":"可读","paper_type_zh":"多轮工具调用轨迹数据发布与构造配方","best_for_zh":"研究环境型 agent trajectory、模拟用户、成功筛选、SFT 数据构造与发布审计的读者","confidence":"high","one_line":["APIGen-MT releases 5,000 success-filtered Retail/Airline tool-use dialogues from a blueprint-and-simulation pipeline, but omits failed runs, replay state, verifier logs, and the complete xLAM training mixture.","APIGen-MT 发布 gated 的 5,000 条 Retail/Airline 成功轨迹，用蓝图验证与模拟交互构造 SFT 数据，但未公开失败运行、重放状态、验证日志或完整 xLAM 训练混合。"],"why":"It connects API schemas, policies, latent environment state, mixed validation, simulated users, and assistant-token behavioral cloning in one auditable construction chain, while showing why a conversation-only success subset cannot by itself reproduce the environment or establish unrestricted training reuse.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/hash/5e3661f7fe4c8ac5652d62eb3d3c96ea-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Salesforce/APIGen-MT-5k"},{"key":"project","label":["Project","项目主页"],"url":"https://apigen-mt.github.io/"}],"link_count":11,"sections":9},{"id":"arc-agi-2-2025","title":"ARC-AGI-2: A New Challenge for Frontier AI Reasoning Systems","year":2025,"venue":"arXiv preprint","authors":["Francois Chollet","Mike Knoop","Gregory Kamradt","Bryan Landers","Henry Pinkard"],"authors_zh":"Francois Chollet 等（ARC Prize）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["abstract-reasoning","grid-transformation","fluid-intelligence-benchmark"],"tags":["benchmark","abstract-reasoning","grid-transformation","exact-match"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"关注抽象推理、少样例规则归纳、exact verifier 和 public/private benchmark 泄漏边界的读者。","confidence":"high","one_line":["ARC-AGI-2 evaluates few-shot abstract grid transformation with exact output-grid scoring.","ARC-AGI-2 用少样例彩色网格转换任务评测抽象规则归纳，并以输出网格 exact match 验收。"],"why":"It provides a high-signal benchmark for abstraction, rule induction, and public/private leakage-aware evaluation.","primary_link":"https://arxiv.org/abs/2505.11831","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/arcprize/ARC-AGI-2"},{"key":"data","label":["Data","数据"],"url":"https://github.com/arcprize/ARC-AGI-2/tree/main/data"},{"key":"project","label":["Project","项目主页"],"url":"https://arcprize.org/"}],"link_count":5,"sections":9},{"id":"ember-llm-judge-uncertainty-2025","title":"Are LLM-Judges Robust to Expressions of Uncertainty? Investigating the effect of Epistemic Markers on LLM-based Evaluation","year":2025,"venue":"NAACL 2025","authors":["Dongryeol Lee","Yering Hwang","Yongil Kim","Joonsuk Park","Kyomin Jung"],"authors_zh":"Dongryeol Lee, Yering Hwang, Yongil Kim, Joonsuk Park, Kyomin Jung","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","reward-model","public-artifact"],"status":"verified","priority":"可读","paper_type_zh":"奖励模型、评估器或验证可靠性审计论文","best_for_zh":"需要检查奖励、评估或偏好信号可靠性的研究者。","confidence":"high","one_line":["EMBER is a released benchmark showing that LLM judges can systematically penalize epistemic markers even when answer correctness is unchanged.","EMBER 证明 LLM 评判器会因不确定性措辞偏离正确性判决，并提供可复用的反事实审计基准。"],"why":"It provides an auditable public surface for measuring or mitigating reward and judge reliability failures.","primary_link":"https://aclanthology.org/2025.naacl-long.452/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DongryeolLee96/EMBER"}],"link_count":3,"sections":9},{"id":"areal-2025","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Wei Fu","Jiaxuan Gao","Xujie Shen","Chen Zhu","Zhiyu Mei","Chuyi He","Shusheng Xu","Guo Wei","Jun Mei","Jiashu Wang","Tongkai Yang","Binhang Yuan","Yi Wu"],"authors_zh":"Wei Fu、Jiaxuan Gao、Xujie Shen、Chen Zhu、Zhiyu Mei、Chuyi He、Shusheng Xu、Guo Wei、Jun Mei、Jiashu Wang、Tongkai Yang、Binhang Yuan、Yi Wu","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning","code_generation","asynchronous_rl_systems"],"tags":["areal","asynchronous-rl","systems","ppo","decoupled-ppo","staleness","rollout","reasoning","programmatic-reward","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"异步强化学习系统、程序化奖励契约与轨迹血缘披露审计","best_for_zh":"关注大规模异步 RL、策略陈旧性、跨版本 rollout、程序化验证器和系统可复现性的研究者与工程人员","confidence":"high","one_line":["AReaL releases Apache-2.0 infrastructure and a NeurIPS asynchronous PPO recipe for math/code RL, but no canonical reasoning corpus, paper-run rollout/reward ledger, immutable data split or final-checkpoint provenance package was verified.","AReaL 发布了用于数学/代码 RL 的 Apache-2.0 异步 PPO 基础设施和 NeurIPS 系统配方；未核验到规范推理语料、论文运行 rollout/奖励账本、不可变数据切分或最终检查点溯源包。"],"why":"It makes asynchronous RL's operational feedback contract inspectable—especially stale and cross-version rollouts—while demonstrating that high-throughput code and reported hyperparameters do not by themselves disclose source rights, reward decisions, trajectory lineage or evaluator validity.","primary_link":"https://arxiv.org/abs/2505.24298","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/areal-project/AReaL"},{"key":"project","label":["Project","项目主页"],"url":"https://areal-project.github.io/AReaL/"}],"link_count":5,"sections":9},{"id":"atlas-autoformalization-data-2025","title":"ATLAS: Autoformalizing Theorems through Lifting, Augmentation, and Synthesis of Data","year":2025,"venue":"NeurIPS 2025","authors":["Xiaoyang Liu","Yicheng Hu","Jinghui Lu","Jian Guo","Yaochu Jin","Jie Zhou","Tao Yang","Mingxuan Wang"],"authors_zh":"Xiaoyang Liu、Yicheng Hu、Jinghui Lu、Jian Guo、Yaochu Jin、Jie Zhou、Tao Yang、Mingxuan Wang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4"],"tags":["autoformalization","lean4","theorem-proving","2025"],"status":"verified","priority":"可读","paper_type_zh":"Lean 自动形式化数据集与验证论文","best_for_zh":"构建或使用 Lean 可检验自动形式化训练数据的研究者。","confidence":"high","one_line":["ATLAS expands natural-language–Lean theorem pairs through lifting, augmentation, and synthesis, filtering autoformalization data with Lean checking.","ATLAS 通过提升、增广和合成扩展自然语言—Lean 定理对，并以 Lean 检查筛选自动形式化数据。"],"why":"ATLAS expands natural-language–Lean theorem pairs through lifting, augmentation, and synthesis, filtering autoformalization data with Lean checking.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/hash/5159aaee380391c366b27994ed225e4f-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XiaoyangLiu-sjtu/ATLAS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/XiaoyangLiu-sjtu/ATLAS_dataset"}],"link_count":3,"sections":9},{"id":"atom-of-thoughts-tts-2025","title":"Atom of Thoughts for Markov LLM Test-Time Scaling","year":2025,"venue":"NeurIPS 2025","authors":["Fengwei Teng","Quan Shi","Zhaoyang Yu","Jiayi Zhang","Yuyu Luo","Chenglin Wu","Zhijiang Guo"],"authors_zh":"Fengwei Teng、Quan Shi、Zhaoyang Yu、Jiayi Zhang、Yuyu Luo、Chenglin Wu、Zhijiang Guo（机构：香港科技大学（广州）、DeepWisdom、中国人民大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","software-engineering","multi-hop-reasoning"],"tags":["test-time-compute","markov-reasoning","decomposition","context-efficiency","search"],"status":"verified","priority":"可读","paper_type_zh":"测试时计算扩展框架研究（NeurIPS 2025）","best_for_zh":"希望降低多步推理历史上下文开销、并结合搜索或反思机制的读者。","confidence":"high","one_line":["Atom of Thoughts converts a problem into answer-equivalent atomic states so inference compute targets the current subproblem rather than an accumulated history.","Atom of Thoughts 将复杂问题转为答案等价的原子状态，使测试时计算聚焦当前子问题而非累积历史。"],"why":"It makes the state representation and context consumption explicit parts of the test-time compute decision.","primary_link":"https://arxiv.org/abs/2502.12018","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/qixucen/atom"}],"link_count":4,"sections":9},{"id":"acpo-atomic-consistency-preference-2025","title":"Atomic Consistency Preference Optimization for Long-Form Question Answering","year":2025,"venue":"IJCNLP-AACL 2025","authors":["Jingfeng Chen","Raghuveer Thirukovalluru","Junlin Wang","Kaiwei Luo","Bhuwan Dhingra"],"authors_zh":"Jingfeng Chen、Raghuveer Thirukovalluru、Junlin Wang、Kaiwei Luo、Bhuwan Dhingra（杜克大学昆山大学、杜克大学、TeleAI）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["long-form-question-answering","factuality"],"tags":["factuality","self-supervision","preference-data","dpo","long-form-generation"],"status":"verified","priority":"必读","paper_type_zh":"自监督偏好数据构造与事实性对齐研究","best_for_zh":"适合研究如何在不依赖外部评审器或知识库的条件下，把模型重复输出转化为偏好监督的读者。","confidence":"high","one_line":["ACPO turns agreement among sentence-level facts in sampled long-form answers into chosen-rejected pairs for factuality-oriented DPO.","ACPO 将长回答中句子级事实在多次采样中的一致程度转化为优选—劣选回答对，并用它进行面向事实性的直接偏好优化。"],"why":"It makes atomic agreement, cluster frequency, and response ranking explicit data-construction decisions rather than relying on a stronger factuality judge.","primary_link":"https://aclanthology.org/2025.ijcnlp-long.106/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JingfengSteven/ACPO"}],"link_count":5,"sections":9},{"id":"audioskills-xl-2025","title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","year":2025,"venue":"arXiv preprint","authors":["Arushi Goel","Sreyan Ghosh","Jaehyeon Kim","Sonal Kumar","Zhifeng Kong","Sang-gil Lee","Chao-Han Huck Yang","Ramani Duraiswami","Dinesh Manocha","Rafael Valle","Bryan Catanzaro"],"authors_zh":"Arushi Goel、Sreyan Ghosh、Jaehyeon Kim、Sonal Kumar、Zhifeng Kong、Sang-gil Lee、Chao-Han Huck Yang、Ramani Duraiswami、Dinesh Manocha、Rafael Valle、Bryan Catanzaro","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["large-scale-open-audio-instruction-and-reasoning-data"],"tags":["instruction-demonstration-rationale","arxiv-2507.08128","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"音频语言监督微调与推理调优","confidence":"high","one_line":["AudioSkills-XL expands multiple audio sources into 8M capability-labeled QA pairs and separates a 250K controlled-thought subset for reasoning-focused training.","AudioSkills-XL 把声音、音乐和语音来源扩展为 800 万条问答，并另设 25 万条受控思考记录。"],"why":"Audio-language models lack a large openly documented instruction mixture spanning sound, music, speech, and reasoning instead of captioning alone.","primary_link":"https://arxiv.org/abs/2507.08128","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/AudioSkills"}],"link_count":2,"sections":9},{"id":"setupagent-swee-swa-bench-2025","title":"Automated Benchmark Generation for Repository-Level Coding Tasks","year":2025,"venue":"ICML 2025","authors":["Konstantinos Vergopoulos","Mark Niklas Müller","Martin Vechev"],"authors_zh":"Konstantinos Vergopoulos, Mark Niklas Müller, Martin Vechev","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","construction_recipe"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software_engineering","repository_level_code","code_agents","python"],"tags":["environment-agent-trajectory-data","software-engineering-agent","repository-level-benchmark","setupagent","SWA-Bench","SWEE-Bench","executable-tests","historical-environment","benchmark-generation"],"status":"partial","priority":"可读","paper_type_zh":"仓库级代码智能体基准与自动构造配方","best_for_zh":"研究仓库级代码智能体评测、历史环境重建、测试终态判定和基准发布审计的读者","confidence":"high","one_line":["SetUpAgent turns issue-resolving GitHub pull requests into executable repository tasks by reconstructing historical dependencies and parsing tests; paper-time SWA/SWEE scales are 535/885 tasks, while current Hub releases expose 450/798.","SetUpAgent 通过历史依赖重建、测试解析与 F2P/P2P 终态契约构造 SWA-Bench 和 SWEE-Bench，但公开行数漂移、数据集许可和生成源码仍阻碍可复现实验。"],"why":"It shows how an agent benchmark's real data object includes pinned repository state, setup commands, tests, parsers, containers, and a terminal predicate, and why release drift can undermine otherwise programmatic evaluation.","primary_link":"https://proceedings.mlr.press/v267/vergopoulos25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/logic-star-ai/SWEBench/tree/c11e96679247719201db83c06012adb3221b33be"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LogicStar/SWA-Bench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/LogicStar/SWEE-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://www.sri.inf.ethz.ch/publications/mueller2025automated"}],"link_count":10,"sections":9},{"id":"autods-mathematical-text-selection-2025","title":"Autonomous Data Selection with Zero-shot Generative Classifiers for Mathematical Texts","year":2025,"venue":"Findings of ACL 2025","authors":["Yifan Zhang","Yifan Luo","Yang Yuan","Andrew C. Yao"],"authors_zh":"Yifan Zhang, Yifan Luo, Yang Yuan, Andrew C. Yao","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["unknown"],"construction_layer":["prompt_sourcing","optimizer_scaffold"],"domains":["mathematical-reasoning","continual-pretraining"],"tags":["post-training","training-usage","data-selection"],"status":"verified","priority":"可读","paper_type_zh":"数学文本筛选与持续预训练数据选择论文（Findings of ACL 2025）","best_for_zh":"研究数学语料筛选、持续预训练和模型相对数据质量的读者。","confidence":"high","one_line":["AutoDS ranks mathematical text with zero-shot generative-classifier logits before continual pretraining.","AutoDS 用零样本生成式分类器词元概率给数学文本排序，再决定哪些数据进入持续预训练。"],"why":"It makes the connection between a data object and its training objective inspectable.","primary_link":"https://aclanthology.org/2025.findings-acl.216/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yifanzhang-pro/AutoMathText"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/math-ai/AutoMathText"}],"link_count":5,"sections":9},{"id":"amazon-nova-2-lite-service-card-2025","title":"AWS AI Service Cards: Amazon Nova 2 Lite","year":2025,"venue":"AWS AI Service Card","authors":["Amazon Web Services"],"authors_zh":"Amazon Web Services","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation","audit","test_time_compute"],"construction_layer":["prompt_sourcing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","multimodal","agentic_tool_use","coding","safety"],"tags":["amazon","aws","nova-2-lite","service-card","frontier-report","data-disclosure-ledger","sft","rlhf","safety-evaluation","multimodal"],"status":"partial","priority":"必读","paper_type_zh":"云服务模型卡与数据披露台账","best_for_zh":"审查云端前沿模型的数据、反馈、运行时安全和审计边界的读者","confidence":"high","one_line":["AWS's Nova 2 Lite Service Card discloses high-level source categories, SFT/RLHF, filters, and evaluation layers but withholds records, rewards, and reproducibility artifacts.","AWS 的 Nova 2 Lite 服务卡披露了高层来源类别、SFT/RLHF、过滤和评测层，但没有公开记录、奖励或可复现制品。"],"why":"A cloud-provider frontier disclosure baseline that separates named post-training and safety interfaces from unavailable provenance, feedback, reward, trace, and audit artifacts.","primary_link":"https://docs.aws.amazon.com/pdfs/ai/responsible-ai/nova-2-lite/nova-2-lite.pdf","links":[],"link_count":2,"sections":9},{"id":"badjudge-2025","title":"BadJudge: Backdoor Vulnerabilities of LLM-As-A-Judge","year":2025,"venue":"ICLR 2025 (Poster)","authors":["Terry Tong","Fei Wang","Zhe Zhao","Muhao Chen"],"authors_zh":"Terry Tong, Fei Wang, Zhe Zhao, Muhao Chen","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"LLM 判别器后门攻击与防御审计","best_for_zh":"构建、微调或部署自动评测器、奖励模型、安全护栏与检索重排器的研究者。","confidence":"high","one_line":["Defines data-centric backdoors against LLM judges and evaluates model merging as a mitigation.","定义针对 LLM 判别器的数据后门，并以模型合并缓解被操纵的评测结果。"],"why":"It shows that a compromised judge can distort model selection while retaining plausible clean behavior.","primary_link":"https://openreview.net/forum?id=eC2a2IndIt","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TerryTong-Git/badjudge"}],"link_count":3,"sections":9},{"id":"static-to-dynamic-llm-eval-2025","title":"Benchmarking Large Language Models Under Data Contamination: A Survey from Static to Dynamic Evaluation","year":2025,"venue":"EMNLP 2025","authors":["Simin Chen","Yiming Chen","Zexin Li","Yifan Jiang","Zhongwei Wan","Yixin He","Dezhi Ran","Tianle Gu","Haizhou Li","Tao Xie","Baishakhi Ray"],"authors_zh":"Simin Chen，Yiming Chen，Zexin Li，Yifan Jiang，Zhongwei Wan，Yixin He，Dezhi Ran，Tianle Gu，Haizhou Li，Tao Xie，Baishakhi Ray。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["contamination","verifier-audit","evaluation-reliability"],"tags":["track13","audit","2025-2026"],"status":"verified","priority":"必读","paper_type_zh":"数据污染下 LLM 基准的 survey 与评测框架","best_for_zh":"构建、选择或审计抗污染 LLM benchmark 的研究者。","confidence":"high","one_line":["This survey and living repository map static and dynamic defenses against benchmark contamination.","系统梳理污染下静态与动态 LLM 基准，并提出动态基准的六项审计准则。"],"why":"It supplies an audit taxonomy and a maintained artifact index for deciding whether an evaluation can still be trusted.","primary_link":"https://aclanthology.org/2025.emnlp-main.511/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SeekingDream/Static-to-Dynamic-LLMEval"}],"link_count":2,"sections":9},{"id":"chain-of-x-survey-2025","title":"Beyond Chain-of-Thought: A Survey of Chain-of-X Paradigms for LLMs","year":2025,"venue":"COLING 2025","authors":["Yu Xia","Rui Wang","Xu Liu","Mingyan Li","Tong Yu","Xiang Chen","Julian McAuley","Shuai Li"],"authors_zh":"Yu Xia 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["trace_writing"],"domains":["chain-of-thought","reasoning-traces","prompting","agents"],"tags":["foundations-and-primers","chain-of-thought","reasoning-traces","coling-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"Chain-of-X 方法综述","best_for_zh":"已了解思维链、希望认识其扩展形式和适用任务的读者。","confidence":"high","one_line":["A COLING 2025 survey that classifies Chain-of-X methods by their intermediate nodes and their application tasks.","从中间节点类型和应用任务两条线系统整理 Chain-of-X 方法。"],"why":"It helps readers distinguish a chain's representation from the task for which that chain is used.","primary_link":"https://aclanthology.org/2025.coling-main.719/","links":[],"link_count":1,"sections":9},{"id":"lager-internal-representations-2025","title":"Beyond the Surface: Enhancing LLM-as-a-Judge Alignment with Human via Internal Representations","year":2025,"venue":"NeurIPS 2025 (Poster)","authors":["Peng Lai","Jianjie Zheng","Sijie Cheng","Yun Chen","Peng Li","Yang Liu","Guanhua Chen"],"authors_zh":"Peng Lai, Jianjie Zheng, Sijie Cheng, Yun Chen, Peng Li, Yang Liu, Guanhua Chen","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["llm-as-a-judge","internal-representations","audit"],"status":"verified","priority":"可读","paper_type_zh":"评测器、验证器、奖励或基准可靠性研究","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Aggregates score-token logits across frozen transformer layers to improve point-wise judge alignment with humans.","聚合冻结 Transformer 各层的分数 token logit，以提升点式 LLM 评审与人工评分的一致性。"],"why":"It makes the evaluator readout itself an auditable source of misalignment rather than treating final-token scores as fixed.","primary_link":"https://arxiv.org/abs/2508.03550","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sustech-nlp/LAGER"}],"link_count":4,"sections":9},{"id":"self-preference-llm-judgments-2025","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","year":2025,"venue":"EMNLP 2025","authors":["Zhi-Yuan Chen","Hao Wang","Xinyu Zhang","Enrui Hu","Yankai Lin"],"authors_zh":"Zhi-Yuan Chen, Hao Wang, Xinyu Zhang, Enrui Hu, Yankai Lin","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器失效或奖励投机审计论文","best_for_zh":"需要审计基准污染、评估偏差或奖励投机风险的研究者。","confidence":"high","one_line":["gold-judgment-controlled self-preference bias measurement","用 gold judgment 校正回答质量后，测量 LLM judge 的真实自偏好。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://aclanthology.org/2025.emnlp-main.86/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhiyuanc2001/self-preference"}],"link_count":2,"sections":9},{"id":"big-math-large-scale-high-quality-math-rl-2025","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","year":2025,"venue":"arXiv","authors":["Alon Albalak","Duy Phung","Nathan Lile","Rafael Rafailov","Kanishk Gandhi","Louis Castricato","Anikait Singh","Chase Blagden","Violet Xiang","Dakota Mahan","Nick Haber"],"authors_zh":"Alon Albalak, Duy Phung, Nathan Lile, Rafael Rafailov, Kanishk Gandhi, Louis Castricato, Anikait Singh, Chase Blagden, Violet Xiang, Dakota Mahan, Nick Haber","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["mathematics","reasoning"],"tags":["mathematics","rlvr","answer-verification","2025"],"status":"verified","priority":"必读","paper_type_zh":"数学 RLVR 数据集与清洗重构方法","best_for_zh":"需要大规模、可规则核验的数学 RLVR 提示数据的研究者。","confidence":"high","one_line":["Big-Math turns diverse open mathematical sources into a 251,122-item RL dataset of open-ended questions with uniquely checkable closed-form answers.","Big-Math 将多源开放数学题清洗并重构为 251,122 条开放式、闭式且答案唯一可核验的 RL 训练记录。"],"why":"It makes a final-answer verification contract available at scale while preserving source and difficulty metadata.","primary_link":"https://arxiv.org/abs/2502.17387","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SynthLabsAI/big-math"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SynthLabsAI/Big-Math-RL-Verified"}],"link_count":4,"sections":9},{"id":"bigcodearena-reliable-human-preferences-code-generation-2025","title":"BigCodeArena: Unveiling More Reliable Human Preferences in Code Generation via Execution","year":2025,"venue":"arXiv","authors":["Terry Yue Zhuo","Xiaolong Jin","Hange Liu","Juyong Jiang","Tianyang Liu","Chen Gong","Bhupesh Bishnoi","Vaisakhi Mishra","Marek Suppa","Noah Ziems","Saiteja Utpala","Ming Xu","Guangyu Song","Kaixin Li","Yuhan Cao","Bo Liu","Zheng Liu","Sabina Abdurakhmanova","Wenhao Yu","Mengzhao Jia","Jihan Yao","Kenneth Hamilton","Kumar Shridhar","Minh Chien Vu","Dingmin Wang","Jiawei Liu","Zijian Wang","Qian Liu","Binyuan Hui","Meg Risdal","Ahsen Khaliq","Atin Sood","Zhenchang Xing","Wasi Uddin Ahmad","John Grundy","David Lo","Banghua Zhu","Xiaoning Du","Torsten Scholak","Leandro von Werra"],"authors_zh":"Terry Yue Zhuo、Xiaolong Jin、Hange Liu 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","preference-learning"],"tags":["preference-data","code","execution"],"status":"verified","priority":"可读","paper_type_zh":"执行增强的多轮代码偏好数据集、奖励模型评测与基准","best_for_zh":"训练或审计代码奖励模型、代码偏好优化和执行感知评测的研究者","confidence":"high","one_line":["BigCodeArena couples human code preferences with execution to make code-feedback labels more reliable.","BigCodeArena 把可交互的代码执行结果接入人类两两偏好收集，使代码奖励模型能同时利用行为证据与用户判断。"],"why":"It joins subjective feedback to a programmatic correctness surface.","primary_link":"https://arxiv.org/abs/2510.08697","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bigcode-project/bigcodearena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/bigcode/bigcodearena-preference-5k"}],"link_count":4,"sections":9},{"id":"bigcodebench-diverse-function-calls-2025","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","year":2025,"venue":"ICLR 2025","authors":["Terry Yue Zhuo","Minh Chien Vu","Jenny Chim","Han Hu","Wenhao Yu","Ratnadira Widyasari","Imam Nur Bani Yusuf","Haolan Zhan","Junda He","Indraneil Paul","Simon Brunner","Chen Gong","Thong Hoang","Armel Zebaze","Xiaoheng Hong","Wen-Ding Li","Jean Kaddour","Ming Xu","Zhihan Zhang","Prateek Yadav","Naman Jain","Alex Gu","Zhoujun Cheng","Jiawei Liu","Qian Liu","Zijian Wang","Binyuan Hui","Niklas Muennighoff","David Lo","Daniel Fried","Xiaoning Du","Harm de Vries","Leandro von Werra"],"authors_zh":"Terry Yue Zhuo、Minh Chien Vu、Jenny Chim、Han Hu、Wenhao Yu、Ratnadira Widyasari、Imam Nur Bani Yusuf、Haolan Zhan、Junda He、Indraneil Paul、Simon Brunner、Chen Gong、Thong Hoang、Armel Zebaze、Xiaoheng Hong、Wen-Ding Li、Jean Kaddour、Ming Xu、Zhihan Zhang、Prateek Yadav、Naman Jain、Alex Gu、Zhoujun Cheng、Jiawei Liu、Qian Liu、Zijian Wang、Binyuan Hui、Niklas Muennighoff、David Lo、Daniel Fried、Xiaoning Du、Harm de Vries、Leandro von Werra","tracks":["programmatically_verifiable_outcome_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","python","library-use"],"tags":["code-generation","function-calls","unit-tests","2025"],"status":"verified","priority":"必读","paper_type_zh":"复杂函数调用代码生成基准","best_for_zh":"研究实用代码生成、库调用、复杂指令与单元测试评测的研究者。","confidence":"high","one_line":["BigCodeBench uses high-coverage tests to assess executable Python generation under diverse library calls and complex instructions.","BigCodeBench 用高覆盖单元测试考察模型在多库函数调用与复杂指令下生成可执行 Python 程序的能力。"],"why":"BigCodeBench uses high-coverage tests to assess executable Python generation under diverse library calls and complex instructions.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/a6a90bcc2aa470c3871b2d39a67d26e8-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bigcode-project/bigcodebench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/bigcode/bigcodebench"},{"key":"project","label":["Project","项目主页"],"url":"https://bigcode-bench.github.io/"}],"link_count":6,"sections":9},{"id":"blackbox-model-provenance-2025","title":"Blackbox Model Provenance via Palimpsestic Membership Inference","year":2025,"venue":"NeurIPS 2025","authors":["Rohith Kuditipudi","Jing Huang","Sally Zhu","Diyi Yang","Christopher Potts","Percy Liang"],"authors_zh":"Rohith Kuditipudi, Jing Huang, Sally Zhu, Diyi Yang, Christopher Potts, Percy Liang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"黑盒模型来源与训练谱系统计审计论文","best_for_zh":"需要审计模型衍生关系、训练谱系与来源证据的研究者。","confidence":"high","one_line":["model-provenance membership-inference data and code","以训练样本随机顺序的相关性，为黑盒衍生模型或文本提供可量化来源证据。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://arxiv.org/abs/2510.19796","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RohithKuditipudi/blackbox-model-tracing"}],"link_count":3,"sections":9},{"id":"mm-detect-multimodal-contamination-2025","title":"Both Text and Images Leaked! A Systematic Analysis of Data Contamination in Multimodal LLM","year":2025,"venue":"Findings of EMNLP 2025","authors":["Dingjie Song","Sicheng Lai","Mingxuan Wang","Shunian Chen","Lichao Sun","Benyou Wang"],"authors_zh":"Dingjie Song, Sicheng Lai, Mingxuan Wang, Shunian Chen, Lichao Sun, Benyou Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要测量多模态基准泄漏及模型评测可信度的研究者。","confidence":"medium","one_line":["Accepted multimodal contamination audit with the authors’ MM-Detect public repository.","将文本与图像泄漏统一为多模态污染审计，并提供可复现的来源分析框架。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.findings-emnlp.556/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MLLM-Data-Contamination/MM-Detect"}],"link_count":2,"sections":9},{"id":"bpp-search-2025","title":"BPP-Search: Enhancing Tree of Thought Reasoning for Mathematical Modeling Problem Solving","year":2025,"venue":"ACL 2025","authors":["Teng Wang","Wing-Yin Yu","Zhenqi He","Zehua Liu","Hailei Gong","Han Wu","Xiongwei Han","Wei Shi","Ruifeng She","Fangzhou Zhu","Tao Zhong"],"authors_zh":"Teng Wang、Wing-Yin Yu、Zhenqi He、Zehua Liu、Hailei Gong、Han Wu、Xiongwei Han、Wei Shi、Ruifeng She、Fangzhou Zhu、Tao Zhong","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["data_release","process_supervision","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["answer_level","step_level","pairwise_preference","process_reward"],"training_use":["sft","preference_learning","reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["operations_research","mathematical_modeling","optimization","industrial_decision_making"],"tags":["bpp-search","structuredor","tree-of-thought","beam-search","process-reward-model","pairwise-preference","operations-research","structured-modeling","final-example-release","raw-tree-not-released"],"status":"partial","priority":"必读","paper_type_zh":"树搜索方法与最终结构化数据集发布","best_for_zh":"研究测试时树搜索、过程奖励模型、候选选择器、运筹优化推理数据与发布审计的读者","confidence":"high","one_line":["BPP-Search uses a learned process reward model to prune a mathematical-modeling tree and a pairwise classifier to select the final candidate, while the public StructuredOR artifact contains 124 final structured examples rather than the raw search trees.","BPP-Search 用学习式过程奖励模型剪枝数学建模树，再用成对偏好分类器选择最终候选；公开的 StructuredOR 仅含 124 条最终结构化样例，而不含原始搜索树。"],"why":"It makes tree width, intermediate scoring, and final-candidate selection explicit data-construction variables, but also demonstrates why a final question and label release cannot support audits of rejected branches, selector behavior, verifier outputs, or inference-budget attribution.","primary_link":"https://aclanthology.org/2025.acl-long.40/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LLM4OR/StructuredOR"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LLM4OR/StructuredOR"}],"link_count":6,"sections":9},{"id":"self-evolution-reasoning-survey-2025","title":"Breaking the Reasoning Barrier A Survey on LLM Complex Reasoning through the Lens of Self-Evolution","year":2025,"venue":"Findings of ACL 2025","authors":["Tao He","Hao Li","Jingchang Chen","Runxuan Liu","Yixin Cao","Lizi Liao","Zihao Zheng","Zheng Chu","Jiafeng Liang","Ming Liu","Bing Qin"],"authors_zh":"Tao He 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","reasoning-data","self-evolution","reinforcement-learning"],"tags":["foundations-and-primers","reasoning-data","self-evolution","acl-2025","survey"],"status":"verified","priority":"必读","paper_type_zh":"复杂推理自我演化综述","best_for_zh":"设计合成推理数据流程或研究迭代式模型改进的读者。","confidence":"high","one_line":["An ACL survey of growing reasoning ability by jointly improving training data and models.","从数据演化、模型演化和循环改进理解复杂推理能力的 ACL 综述。"],"why":"It gives readers a compact way to reason about where iterative reasoning systems obtain data, quality signals, and real improvements.","primary_link":"https://aclanthology.org/2025.findings-acl.386/","links":[],"link_count":2,"sections":9},{"id":"bridging-provenance-gap-2025","title":"Bridging the Data Provenance Gap Across Text, Speech, and Video","year":2025,"venue":"ICLR","authors":["Shayne Longpre","Nikhil Singh","Manuel Cherep","Kushagra Tiwary","Joanna Materzynska","William Brannon","Robert Mahari","Naana Obeng-Marnu","Manan Dey","Mohammed Hamdy","Nayan Saxena","Ahmad Mustafa Anis","Emad A. Alghamdi","Vu Minh Chien","Da Yin","Kun Qian","Yizhi Li","Minnie Liang","An Dinh","Shrestha Mohanty","Deividas Mataciunas","Tobin South","Jianguo Zhang","Ariel N. Lee","Campbell S. Lund","Christopher Klamm","Damien Sileo","Diganta Misra","Enrico Shippole","Kevin Klyman","Lester JV Miranda","Niklas Muennighoff","Seonghyeon Ye","Seungone Kim","Vipul Gupta","Vivek Sharma","Xuhui Zhou","Caiming Xiong","Luis Villa","Stella Biderman","Alex Pentland","Sara Hooker","Jad Kabbara"],"authors_zh":"Shayne Longpre 等 43 位作者","tracks":["data_construction_open_release_recipes"],"source_role":["audit_failure","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["unknown"],"training_use":["audit"],"construction_layer":["prompt_sourcing","release_audit","frontier_pipeline"],"domains":["general_reasoning","multimodal"],"tags":["data-provenance","dataset-licensing","source-terms","multimodal-audit","derivation-lineage","human-metadata-audit","release-versioning","data-governance"],"status":"verified","priority":"可读","paper_type_zh":"多模态数据 provenance、许可证与来源条款审计","best_for_zh":"负责后训练数据选型、开放发布、权利复核、来源治理与版本审计的研究者和工程团队","confidence":"medium","one_line":["Releases a human-audited dataset-level catalogue covering 3,916 reported text, speech, and video datasets, with source, derivation, creator, representation, license, and source-terms metadata but no per-record provenance.","发布人工审计的数据集级目录，覆盖论文表中 3,916 个文本、语音与视频数据集的来源、派生、创建者、表示、许可证和来源条款，但不提供逐记录 provenance。"],"why":"For the construction/open-release track, it provides an operational rights and lineage review layer before composing reasoning-data mixtures, and shows that permissive dataset labels can conflict with upstream source restrictions.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/97b983c974551153d20ddfabb62a5203-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Data-Provenance-Initiative/Data-Provenance-Collection"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/DataProvenanceInitiative/datasets"},{"key":"project","label":["Project","项目主页"],"url":"https://www.dataprovenance.org/"}],"link_count":8,"sections":9},{"id":"browsecomp-2025","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","year":2025,"venue":"arXiv preprint (arXiv:2504.12516)","authors":["Jason Wei","Zhiqing Sun","Spencer Papay","Scott McKinney","Jeffrey Han","Isa Fulford","Hyung Won Chung","Alex Tachard Passos","William Fedus","Amelia Glaese"],"authors_zh":"Jason Wei, Zhiqing Sun, Spencer Papay, Scott McKinney, Jeffrey Han, Isa Fulford, Hyung Won Chung, Alex Tachard Passos, William Fedus, Amelia Glaese","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit","test_time_compute"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["web_agents","environment_interaction","live_web","information_seeking","factual_qa"],"tags":["browsecomp","live-web","web-agent","information-seeking","short-answer-benchmark","llm-answer-judge","test-time-compute","reversible-obfuscation","canary","contamination-risk","answer-only-feedback","no-trajectory-release","non-replayable-environment","evaluator-bug"],"status":"partial","priority":"可读","paper_type_zh":"网页智能体基准与数据发布","best_for_zh":"研究实时网页智能体评测、答案级 judge、测试时计算与污染审计的读者","confidence":"medium","one_line":["BrowseComp releases 1,266 obfuscated hard factual web questions and short answers scored by an answer-only LLM judge, but no browsing trajectories, citations, or replayable web state, and its current official reference scorer has a label-parsing defect.","BrowseComp 发布 1,266 道经可逆混淆的高难度网页事实问答题，以答案级 LLM judge 评分；它不发布浏览轨迹、引用或可回放网页状态，且当前官方参考 scorer 存在标签解析缺陷。"],"why":"It is a canonical stress test for persistent live-web search and test-time compute, while its reversible anti-leakage layer, single-reference judge, undisclosed paper grader, post-release contamination risk, missing episodes, and scoring-code defect show why benchmark availability is not the same as trajectory reuse or reproducible feedback.","primary_link":"https://arxiv.org/abs/2504.12516","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/simple-evals"},{"key":"data","label":["Data","数据"],"url":"https://openaipublic.blob.core.windows.net/simple-evals/browse_comp_test_set.csv"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/browsecomp/"}],"link_count":7,"sections":9},{"id":"buildbench-2025","title":"BuildBench: Benchmarking LLM Agents on Compiling Real-World Open-Source Software","year":2025,"venue":"NeurIPS 2025 Deep Learning for Code Workshop (DL4C), Poster","authors":["Zehua Zhang","Ati Priya Bajaj","Divij Handa","Siyu Liu","Arvind S Raj","Hongkai Chen","Hulin Wang","Yibo Liu","Zion Leonahenahe Basque","Souradip Nath","Vishal Juneja","Nikhil Chapre","Yan Shoshitaishvili","Adam Doupé","Chitta Baral","Ruoyu Wang"],"authors_zh":"Zehua Zhang, Ati Priya Bajaj, Divij Handa, Siyu Liu, Arvind S Raj, Hongkai Chen, Hulin Wang, Yibo Liu, Zion Leonahenahe Basque, Souradip Nath, Vishal Juneja, Nikhil Chapre, Yan Shoshitaishvili, Adam Doupé, Chitta Baral, Ruoyu Wang","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software_engineering","repository_level_code","code_agents","c_cpp","build_automation","environment_interaction"],"tags":["environment-agent-trajectory-data","software-engineering-agent","repository-compilation","build-benchmark","executable-environment","terminal-predicate","documentation-retrieval","failure-analysis","release-version-drift"],"status":"partial","priority":"可读","paper_type_zh":"环境型仓库编译智能体评测基准","best_for_zh":"研究软件工程 agent、可执行环境反馈、terminal predicate、benchmark replay 与发布审计的读者","confidence":"high","one_line":["BuildBench turns real C/C++ repositories into executable build-agent tasks with README/web retrieval, iterative shell feedback, and strict/flexible target-binary checks; the current release retains 385 candidate labels but not full rollouts or a pinned public scaffold.","BuildBench 在全新 Ubuntu 22.04 容器中以文档检索、迭代 Bash 执行反馈和目标二进制文件名 predicate 评测 C/C++ 仓库编译 agent；当前发布保留 385 个候选标签，但不含完整 command-observation rollout。"],"why":"It cleanly separates a repository task, an interactive execution episode, and a terminal verifier, while exposing how mutable dependencies, source patching, filename-only success, missing failure traces, and release drift can undermine reuse of environment-agent data.","primary_link":"https://arxiv.org/abs/2509.25248v1","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/STEVENZHANG904/Build_Bench_Test_Data"}],"link_count":8,"sections":9},{"id":"compute-optimal-tts-model-reward-2025","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","year":2025,"venue":"arXiv preprint","authors":["Runze Liu","Junqi Gao","Jian Zhao","Kaiyan Zhang","Xiu Li","Biqing Qi","Wanli Ouyang","Bowen Zhou"],"authors_zh":"Runze Liu、Junqi Gao、Jian Zhao、Kaiyan Zhang、Xiu Li、Biqing Qi、Wanli Ouyang、Bowen Zhou（机构：上海人工智能实验室、清华大学、哈尔滨工业大学、北京邮电大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","compute-optimal","process-reward-model","search","mathematical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"计算最优测试时扩展的实证研究","best_for_zh":"比较不同模型规模和任务难度下过程奖励模型引导采样与搜索的读者。","confidence":"high","one_line":["The study shows that optimal inference-time search depends jointly on generator, verifier, and problem difficulty rather than model size alone.","研究表明最优推理时搜索由生成器、验证器和题目难度共同决定，而非仅由模型规模决定。"],"why":"It makes reward-model compatibility an explicit part of selecting a practical test-time scaling strategy.","primary_link":"https://arxiv.org/abs/2502.06703","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RyanLiu112/compute-optimal-tts"},{"key":"project","label":["Project","项目主页"],"url":"https://ryanliu112.github.io/compute-optimal-tts/"}],"link_count":4,"sections":9},{"id":"deltabench-long-chain-of-thought-errors-2025","title":"Can Large Language Models Detect Errors in Long Chain-of-Thought Reasoning?","year":2025,"venue":"ACL 2025","authors":["Yancheng He","Shilong Li","Jiaheng Liu","Weixun Wang","Xingyuan Bu","Ge Zhang","Zhongyuan Peng","Zhaoxiang Zhang","Zhicheng Zheng","Wenbo Su","Bo Zheng"],"authors_zh":"Yancheng He, Shilong Li, Jiaheng Liu, Weixun Wang, Xingyuan Bu, Ge Zhang, Zhongyuan Peng, Zhaoxiang Zhang, Zhicheng Zheng, Wenbo Su, Bo Zheng","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","code-reasoning","general-reasoning","evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要诊断长链推理、评测首错定位与过程批评能力的研究者。","confidence":"high","one_line":["DeltaBench pairs 1,236 long o1-like traces with fine-grained annotations to test whether critics can find process errors beyond final-answer accuracy.","DeltaBench 为 1,236 条长推理轨迹提供细粒度过程标注，用于检验批评器能否发现终局答案之外的过程错误。"],"why":"It exposes the failure of short-trace process evaluators on realistic long reasoning and provides a public diagnostic surface.","primary_link":"https://aclanthology.org/2025.acl-long.905/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenStellarTeam/DeltaBench"}],"link_count":4,"sections":9},{"id":"longreps-reasoning-path-supervision-2025","title":"Chain-of-Thought Matters: Improving Long-Context Language Models with Reasoning Path Supervision","year":2025,"venue":"Findings of EMNLP 2025","authors":["Dawei Zhu","Xiyu Wei","Guangxiang Zhao","Wenhao Wu","Haosheng Zou","Junfeng Ran","Xun Wang","Lin Sun","Xiangzheng Zhang","Sujian Li"],"authors_zh":"Dawei Zhu、Xiyu Wei、Guangxiang Zhao、Wenhao Wu、Haosheng Zou、Junfeng Ran、Xun Wang、Lin Sun、Xiangzheng Zhang、Sujian Li","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["long-context-reasoning","process-supervision","question-answering"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要为长上下文问答构造证据路径监督或评估过程监督跨领域泛化的研究者。","confidence":"high","one_line":["LongRePS bootstraps and quality-filters explicit evidence paths for long documents, training language models to reason through context rather than only predict the final answer.","LongRePS 自举并筛选长文档中的显式证据推理路径，训练模型基于上下文完成推理而非只预测最终答案。"],"why":"It makes the intermediate retrieval-and-aggregation path a released supervision object for long-context models.","primary_link":"https://aclanthology.org/2025.findings-emnlp.170/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lemon-prog123/LongRePS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Lemon123prog/Qwen-2.5-7B-LongRePS"}],"link_count":5,"sections":9},{"id":"chartm3-2025","title":"ChartM3: A Multi-Stage Code-Driven Pipeline for Constructing Multi-Dimensional and Multi-Step Visual Reasoning Data in Chart Comprehension","year":2025,"venue":"Findings of EMNLP 2025","authors":["Duo Xu","Hao Cheng","Xin Lin","Zhen Xie","Hao Henry Wang"],"authors_zh":"Duo Xu、Hao Cheng、Xin Lin、Zhen Xie、Hao Henry Wang","tracks":["data_construction_open_release_recipes"],"source_role":["benchmark","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","scalar_reward"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["chart_comprehension","multimodal","data_visualization","business_analytics"],"tags":["chartm3","chart-reasoning","multimodal","code-driven-generation","synthetic-data","rlvr"],"status":"partial","priority":"必读","paper_type_zh":"图表推理数据构建与验证配方","best_for_zh":"关注可执行多模态数据生成、图表推理 SFT/RLVR 及奖励边界审计的研究者","confidence":"medium","one_line":["ChartM3 reports a code-driven pipeline for 141,800 training and 2,871 test Q&A, but execution checks, learned judges, privileged code-output rationales, and unreconciled retained-versus-final counts remain unauditable without an official release.","ChartM3 报告以数据、渲染和分析代码驱动的图表推理构造流程，但未发布可核验的官方语料，执行、学习型 judge 与代码输出型推理仍须分开审计。"],"why":"It shows how to preserve chart/data/code/question lineage for multimodal SFT and final-answer RLVR, while demonstrating that executable construction and downstream gains do not mechanically certify chart visibility, reasoning faithfulness, learned-judge correctness, or record-level provenance.","primary_link":"https://aclanthology.org/2025.findings-emnlp.701/","links":[],"link_count":4,"sections":9},{"id":"chartmimic-2024","title":"ChartMimic: Evaluating LMM's Cross-Modal Reasoning Capability via Chart-to-Code Generation","year":2025,"venue":"ICLR 2025 / arXiv","authors":["Cheng Yang","Chufan Shi","Yaxin Liu","Bo Shui","Junjie Wang","Mohan Jing","Linran Xu","Xinyu Zhu","Siheng Li","Yuxiang Zhang","Gongye Liu","Xiaomei Nie","Deng Cai","Yujiu Yang"],"authors_zh":"Cheng Yang 等（Tsinghua University、Tencent AI Lab）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["chart-to-code-reasoning","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","chart-to-code-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注多模态推理、可执行代码生成、图表重建、评分契约和污染风险的研究者。","confidence":"medium","one_line":["ChartMimic exposes chart-to-code generation and rendered-output comparison as an auditable evaluation surface.","ChartMimic 用 chart-to-code 生成、代码执行和渲染图相似度评测 LMM 的跨模态推理。"],"why":"Combines visual understanding with executable reconstruction.","primary_link":"https://arxiv.org/abs/2406.09961","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ChartMimic/ChartMimic"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ChartMimic/ChartMimic"},{"key":"project","label":["Project","项目主页"],"url":"https://chartmimic.github.io/"}],"link_count":5,"sections":9},{"id":"checklist-engineering-multilingual-judges-2025","title":"Checklist Engineering Empowers Multilingual LLM Judges","year":2025,"venue":"GlobalNLP 2025","authors":["Mohammad Ghiasvand Mohammadkhani","Hamid Beigy"],"authors_zh":"Mohammad Ghiasvand Mohammadkhani, Hamid Beigy","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"多语言 LLM 判别器与动态核对清单","best_for_zh":"需要低成本、可解释的多语言逐点评分或成对比较评测的研究者。","confidence":"high","one_line":["Provides a public multilingual judge method and evaluates checklist-based reliability.","用双向动态核对清单引导开源 7B 模型，在不微调条件下执行多语言文本评测。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.globalnlp-1.21/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mghiasvand1/CE-Judge"}],"link_count":3,"sections":9},{"id":"rlcf-checklists-alignment-2025","title":"Checklists Are Better Than Reward Models For Aligning Language Models","year":2025,"venue":"NeurIPS 2025","authors":["Vijay Viswanathan","Yanchao Sun","Shuang Ma","Xiang Kong","Meng Cao","Graham Neubig","Tongshuang Wu"],"authors_zh":"Vijay Viswanathan、Yanchao Sun、Shuang Ma、Xiang Kong、Meng Cao、Graham Neubig、Tongshuang Wu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["RLCF generates weighted checklists from instructions and uses item-level AI/program judgments as reinforcement-learning feedback.","RLCF 从指令生成加权核对表，以逐项 AI／程序判分构成强化学习反馈，替代固定奖励模型。"],"why":"RLCF generates weighted checklists from instructions and uses item-level AI/program judgments as reinforcement-learning feedback.","primary_link":"https://arxiv.org/abs/2507.18624","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/viswavi/RLCF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/viswavi/rlcf"}],"link_count":3,"sections":9},{"id":"cheems-chinese-reward-models-2025","title":"CHEEMS: A Practical Guidance for Building and Evaluating Chinese Reward Models","year":2025,"venue":"ACL 2025","authors":["Xueru Wen","Jie Lou","Zichao Li","Yaojie Lu","Xing Yu","Yuqiu Ji","Guohai Xu","Hongyu Lin","Ben He","Xianpei Han","Le Sun","Debing Zhang"],"authors_zh":"Xueru Wen, Jie Lou, Zichao Li, Yaojie Lu, Xing Yu, Yuqiu Ji, Guohai Xu, Hongyu Lin, Ben He, Xianpei Han, Le Sun, Debing Zhang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","reward-model","public-artifact"],"status":"verified","priority":"可读","paper_type_zh":"中文奖励模型数据集与评测论文","best_for_zh":"需要构建、评测或审计中文奖励模型的研究者。","confidence":"high","one_line":["A released Chinese reward-model benchmark and preference dataset designed to expose gaps in reward-model training and evaluation.","CHEEMS 构建人工监督的中文奖励模型基准与偏好数据，揭示并改善中文偏好建模缺口。"],"why":"It provides an auditable public surface for measuring or mitigating reward and judge reliability failures.","primary_link":"https://aclanthology.org/2025.acl-long.737/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AlignRM/CheemsRM"}],"link_count":2,"sections":9},{"id":"cheems-chinese-reward-models-acl-2025","title":"Cheems: A Practical Guidance for Building and Evaluating Chinese Reward Models from Scratch","year":2025,"venue":"ACL 2025","authors":["Xueru Wen","Jie Lou","Zichao Li","Yaojie Lu et al."],"authors_zh":"Xueru Wen、Jie Lou、Zichao Li、Yaojie Lu 等","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["preference-feedback-batch-2026","post-training","data-construction"],"status":"verified","priority":"可读","paper_type_zh":"偏好与奖励反馈数据集或数据构建研究","best_for_zh":"需要构建、审计或复用偏好与奖励反馈数据的研究者。","confidence":"high","one_line":["Introduces Cheems, an end-to-end recipe for Chinese reward models covering preference-data construction, reward-model training, and localized evaluation.","提出 Cheems 中文奖励模型构建配方：覆盖偏好数据生产、奖励模型训练和本土化评测，可直接复现中文 Reward Model 流程。"],"why":"It exposes a reusable preference or reward-feedback data surface that requires provenance and bias audit before reuse.","primary_link":"https://aclanthology.org/2025.acl-long.737/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AlignRM/CheemsRM"},{"key":"data","label":["Data","数据"],"url":"https://github.com/AlignRM/CheemsRM/tree/master/data"}],"link_count":3,"sections":9},{"id":"anthropic-claude-3-7-sonnet-system-card-2025","title":"Claude 3.7 Sonnet System Card","year":2025,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["preference_learning","safety_alignment","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","coding","agentic_tool_use"],"tags":["anthropic","claude-3-7-sonnet","system-card","extended-thinking","pairwise-preference","constitutional-ai","reward-hacking","chain-of-thought-faithfulness","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型系统卡与数据披露账本","best_for_zh":"需要审计闭源前沿模型后训练数据、反馈合同、推理预算和可复现性边界的读者","confidence":"high","one_line":["Claude 3.7 Sonnet's system card discloses broad source categories, classifier-derived appropriate-harmlessness preferences, RL-trained extended thinking, and selected monitors, but not the underlying records, rewards, rollout settings, or replay artifacts.","Claude 3.7 Sonnet 系统卡披露了宽泛来源类别、由分类器构造的适当无害性偏好、经强化学习训练的扩展思维及若干监控机制，但未公开底层记录、奖励、采样预算或可回放产物。"],"why":"It lets Track 12 readers separate a concrete pairwise-feedback rule and observable inference controls from undisclosed trace provenance, RL rewards, classifier calibration, scaffolds, and reproducibility evidence.","primary_link":"https://assets.anthropic.com/m/785e231869ea8b3b/original/claude-3-7-sonnet-system-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://www.anthropic.com/news/claude-3-7-sonnet"}],"link_count":2,"sections":9},{"id":"claude-4-system-card-2025","title":"Claude 4 System Card","year":2025,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["preference_learning","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","agentic_tool_use"],"tags":["anthropic","claude-4","frontier-report","data-disclosure-ledger","constitutional-ai","human-feedback","safety"],"status":"partial","priority":"可读","paper_type_zh":"闭源前沿模型系统卡与透明度披露","best_for_zh":"比较闭源前沿模型数据与反馈披露边界的读者","confidence":"high","one_line":["Claude 4's system card and transparency materials name broad source categories, human feedback, and Constitutional AI, but do not disclose record-level data, reward, or environment contracts.","Claude 4 的系统卡和透明度材料列出广义来源类别、人工反馈与 Constitutional AI，但未披露记录级数据、奖励或环境合约。"],"why":"It supplies an independent closed-frontier baseline for distinguishing high-level alignment disclosure from auditable reasoning-data and feedback provenance.","primary_link":"https://www-cdn.anthropic.com/6d8a8055020700718b0c49369f60816ba2a7c285/Claude%204%20System%20Card.pdf","links":[],"link_count":2,"sections":9},{"id":"anthropic-claude-haiku-4-5-system-card-2025","title":"Claude Haiku 4.5 System Card","year":2025,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","state_action_level","full_episode"],"training_use":["sft","preference_learning","agent_training","evaluation","audit","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","computer_use","tool_use","agentic_systems","safety","cybersecurity","biosecurity","chain_of_thought","reward_hacking"],"tags":["anthropic","claude-haiku-4-5","system-card","training-data-disclosure","rlhf","rlaif","preference-data","agentic-rl","context-awareness","supervised-reasoning-traces","chain-of-thought","reward-hacking","training-distribution-evaluation","prompt-injection","recursive-audit","safety-evaluation","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型系统卡与训练数据披露台账","best_for_zh":"研究偏好学习、agentic RL、reasoning trace、reward hacking、安全评测与训练谱系的读者","confidence":"high","one_line":["Anthropic's Haiku 4.5 system card discloses five coarse training-source classes, RLHF/RLAIF, context-aware agent RL, prior-model reasoning traces, and a recursive later-stage training audit, but releases no source proportions, reward contract, global split, or reusable records.","Claude Haiku 4.5 System Card 披露五类训练来源、RLHF/RLAIF、偏好选择、上下文感知 agentic RL、前代模型推理文本与后训练行为审计，但不提供来源比例、reward model、全局 split、授权账本或可复用训练记录。"],"why":"It is unusually useful for a disclosure ledger because it connects source categories, preference labor, reasoning-trace lineage, agentic post-training, and reward-hacking audits in one official artifact, while explicitly revealing that a targeted reward-hacking evaluation comes from the training distribution and therefore should not be mistaken for a clean held-out test.","primary_link":"https://www-cdn.anthropic.com/7aad69bf12627d42234e01ee7c36305dc2f6a970.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://www.anthropic.com/news/claude-haiku-4-5"}],"link_count":3,"sections":9},{"id":"anthropic-claude-opus-4-5-system-card-2025","title":"Claude Opus 4.5 System Card","year":2025,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation","audit","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","agentic_tool_use","safety","cybersecurity","mathematics","vision"],"tags":["anthropic","claude-opus-4-5","claude","frontier-report","system-card","data-disclosure-ledger","rlhf","rlaif","reasoning-text","decontamination","reward-hacking","safety-audit"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露台账","best_for_zh":"需要审查闭源前沿模型中后训练数据、反馈奖励、推理文本监测、去污染和可复现性边界的读者","confidence":"high","one_line":["Claude Opus 4.5's system card discloses broad source categories, RLHF/RLAIF, prior-model reasoning text in earlier supervised learning, no RL reward on reasoning-text content, selected reward-hacking safeguards, and concrete decontamination rules, but not auditable records, rewards, or environments.","Claude Opus 4.5 的系统卡披露了宽泛来源类别、RLHF/RLAIF、早期监督学习中的先前模型 reasoning text、RL 不按 reasoning-text 内容奖惩、部分 reward-hacking 防护和具体去污染规则，但没有公开可审计的记录、奖励或环境。"],"why":"It is a particularly informative Track 12 ledger item because it separates disclosed frontier post-training and monitoring mechanisms from the missing record-level provenance, feedback contract, reward implementation, and reproducibility artifacts.","primary_link":"https://www-cdn.anthropic.com/bf10f64990cfda0ba858290be7b8cc6317685f47.pdf","links":[],"link_count":2,"sections":9},{"id":"anthropic-claude-sonnet-4-5-system-card-2025","title":"Claude Sonnet 4.5 System Card","year":2025,"venue":"Anthropic system card","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation","audit","test_time_compute"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","agentic_tool_use","cybersecurity","mathematics","safety"],"tags":["anthropic","claude-sonnet-4-5","claude","frontier-report","system-card","data-disclosure-ledger","rlhf","rlaif","reward-hacking","mechanistic-interpretability","evaluation-awareness","safety-audit"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露台账","best_for_zh":"需要审查前沿模型报告中训练数据与奖励契约的披露边界，以及 reward hacking、监控和 white-box 安全审计的读者","confidence":"high","one_line":["Claude Sonnet 4.5's system card discloses broad source categories, RLHF/RLAIF, safety-pipeline changes, reward-hacking stress tests, and a white-box evaluation-awareness audit, but not the associated training data, reward functions, model internals, or reproducibility artifacts.","Claude Sonnet 4.5 的系统卡披露了宽泛来源类别、RLHF/RLAIF、安全管线变更、reward-hacking 压力测试和 white-box evaluation-awareness 审计，但没有公开相关训练数据、奖励函数、模型内部或可复现制品。"],"why":"It is a high-value Track 12 report because it makes the difference between disclosed frontier evaluation practice and undisclosed post-training data/reward contracts unusually legible, including a concrete warning that evaluation awareness can confound safety evidence.","primary_link":"https://www-cdn.anthropic.com/963373e433e489a87a10c823c52a0a013e9172dd/Claude%20Sonnet%204.5%20System%20Card.pdf","links":[],"link_count":2,"sections":9},{"id":"prefcleanbench-preference-cleaning-2025","title":"Clean First, Align Later: Benchmarking Preference Data Cleaning for Reliable LLM Alignment","year":2025,"venue":"NeurIPS 2025","authors":["Samuel Yeh","Sharon Li"],"authors_zh":"Samuel Yeh、Sharon Li（威斯康星大学麦迪逊分校）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["release_audit"],"domains":["alignment","safety","instruction-following"],"tags":["data-cleaning","preference-learning","dpo","alignment","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"偏好数据清洗基准与训练使用评测研究","best_for_zh":"适合需要在偏好优化之前选择、修正或删除偏好记录的读者。","confidence":"high","one_line":["PrefCleanBench compares how cleaning preference records changes downstream alignment across data, models, and objectives.","PrefCleanBench 比较偏好记录的不同清洗方式如何影响下游对齐表现。"],"why":"It makes data cleaning a measured training-data decision rather than an untested preprocessing assumption.","primary_link":"https://arxiv.org/abs/2509.23564","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deeplearning-wisc/PrefCleanBench"}],"link_count":3,"sections":9},{"id":"clever-verified-code-generation-2025","title":"CLEVER: A Curated Benchmark for Formally Verified Code Generation","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Amitayush Thakur","Jasper Lee","George Tsoukalas","Meghana Sistla","Matthew Zhao","Stefan Zetzsche","Greg Durrett","Yisong Yue","Swarat Chaudhuri"],"authors_zh":"Amitayush Thakur, Jasper Lee, George Tsoukalas, Meghana Sistla, Matthew Zhao, Stefan Zetzsche, Greg Durrett, Yisong Yue, Swarat Chaudhuri","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-verification","lean4","code-generation"],"tags":["lean4","formal-verification","verified-code","program-synthesis","2025"],"status":"verified","priority":"可读","paper_type_zh":"手工审校的 Lean 形式验证代码生成基准","best_for_zh":"需要联合评测规格生成与可证明 Lean 实现生成的研究者。","confidence":"high","one_line":["CLEVER evaluates end-to-end Lean verified code generation on 161 hand-curated tasks with held-out specifications and post-hoc type checking.","CLEVER 包含 161 个手工审校任务，要求生成与保留规格一致的 Lean 实现，并由类型检查器在事后验证。"],"why":"It avoids test-only shortcuts and checks a complete specification-to-implementation contract in Lean.","primary_link":"https://arxiv.org/abs/2505.13938","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/trishullab/clever"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/amitayusht/clever"}],"link_count":4,"sections":9},{"id":"cm-align-multilingual-preference-2025","title":"CM-Align: Consistency-based Multilingual Alignment for Large Language Models","year":2025,"venue":"Findings of EMNLP 2025","authors":["Xue Zhang","Yunlong Liang","Fandong Meng","Songming Zhang","Yufeng Chen","Jinan Xu","Jie Zhou"],"authors_zh":"Xue Zhang、Yunlong Liang、Fandong Meng 等（北京交通大学、腾讯微信 AI）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["trace_writing"],"domains":["alignment","multilingual","reasoning"],"tags":["multilingual","preference-data-construction","dpo","consistency"],"status":"verified","priority":"可读","paper_type_zh":"跨语言一致性偏好数据构造与多语言对齐研究","best_for_zh":"适合构造多语言偏好数据并用于直接偏好优化的读者。","confidence":"high","one_line":["CM-Align constructs multilingual DPO pairs by selecting a reliable English anchor and cross-lingually consistent responses.","CM-Align 用英语锚点与跨语言一致性构造多语言 DPO 偏好对。"],"why":"It treats the English reference and target-language pair construction as explicit data-quality decisions.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1401/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XZhang00/CM-Align"}],"link_count":3,"sections":9},{"id":"co-evolving-llm-coder-unit-tester-reinforcement-learning-2025","title":"Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning","year":2025,"venue":"NeurIPS 2025 (Spotlight)","authors":["Yinjie Wang","Ling Yang","Ye Tian","Ke Shen","Mengdi Wang"],"authors_zh":"Yinjie Wang, Ling Yang, Ye Tian, Ke Shen, Mengdi Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","self_play_anchor","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["code_generation","unit_testing"],"tags":["cure","reasonflux-coder","co-evolution","unit-tests","code-rl","cross-execution","best-of-n"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"high","one_line":["CURE alternates one policy between coder and tester roles, using a code-by-test execution matrix and private gold-test anchors to train both capabilities and to rank code at test time.","CURE 用代码与单元测试的执行矩阵共同训练生成器和测试器，并在测试时用生成测试选择代码。"],"why":"It makes the feedback lineage of code self-play unusually concrete while exposing the key boundary that no gold code does not mean no gold supervision: hidden tests still define coder correctness and tester rewards.","primary_link":"https://arxiv.org/abs/2506.03136","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Gen-Verse/CURE"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Gen-Verse/CodeContests_train"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/Gen-Verse/reasonflux-coder"}],"link_count":7,"sections":9},{"id":"codeio-2025","title":"CODE I/O: Condensing Reasoning Patterns via Code Input-Output Prediction","year":2025,"venue":"ICML 2025","authors":["Junlong Li","Daya Guo","Dejian Yang","Runxin Xu","Yu Wu","Junxian He"],"authors_zh":"Junlong Li、Daya Guo、Dejian Yang、Runxin Xu、Yu Wu、Junxian He","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["code-reasoning","symbolic-reasoning","scientific-reasoning","logical-reasoning","mathematical-reasoning","commonsense-reasoning"],"tags":["instruction-demonstration-rationale","code-input-output","execution-feedback","self-correction","arxiv-2502.07316","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"执行校验推理示范数据集与两阶段 SFT 配方","best_for_zh":"适合从可执行代码构造广域推理监督，或审计带反馈纠错轨迹的研究者。","confidence":"high","one_line":["CODE I/O transforms 454.9K executable functions into 3.52M CoT prediction records, while publicly releasing the ODC-BY PythonEdu-Reasoning subset.","CODE I/O 把 454.9K 个可执行函数转化为 3.52M 条带程序校验与修订反馈的自然语言推理示范，并公开其中的 PythonEdu-Reasoning 子集。"],"why":"It exposes diverse program logic as natural-language reasoning supervision and makes prediction correctness, feedback, and revision visible in the serialized record.","primary_link":"https://proceedings.mlr.press/v267/li25t.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/CodeIO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hkust-nlp/CodeIO-PyEdu-Reasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://codei-o.github.io/"}],"link_count":6,"sections":9},{"id":"code-migration-2025","title":"Code Migration Benchmark","year":2025,"venue":"Vals AI benchmark page","authors":["Vals AI"],"authors_zh":"Vals AI","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["coding","code-migration"],"tags":["benchmark","code-migration","coding","evaluation-surface"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"关注代码迁移、跨语言重实现、隐藏测试和工程等价性评估的读者。","confidence":"high","one_line":["Code Migration evaluates whether models can reimplement real-world programs in another language and pass hidden behavioral tests in an offline sandbox.","Code Migration 评测模型能否把真实程序迁移到另一种语言，并在离线沙箱中通过隐藏行为测试。"],"why":"Code Migration evaluates whether models can reimplement real-world programs in another language and pass hidden behavioral tests in an offline sandbox.","primary_link":"https://www.vals.ai/benchmarks/code-migration","links":[],"link_count":1,"sections":9},{"id":"code-enhanced-reasoning-survey-2025","title":"Code to Think, Think to Code: A Survey on Code-Enhanced Reasoning and Reasoning-Driven Code Intelligence in LLMs","year":2025,"venue":"EMNLP 2025","authors":["Dayu Yang","Tianyang Liu","Daoan Zhang","Antoine Simoulin","Xiaoyi Liu","Yuwei Cao","Zhaopu Teng","Xin Qian","Grey Yang","Jiebo Luo","Julian McAuley"],"authors_zh":"Dayu Yang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["reward_verifier_layer","trace_writing"],"domains":["code-reasoning","programmatic-verification","agents"],"tags":["foundations-and-primers","code-reasoning","emnlp-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"代码增强推理综述","best_for_zh":"希望结合代码、执行反馈与复杂推理的读者。","confidence":"high","one_line":["An EMNLP 2025 survey of code as a structured, executable medium for reasoning and of reasoning as a driver of code intelligence.","系统梳理代码如何作为可执行的推理媒介，以及推理如何提升代码智能。"],"why":"It clarifies why code can supply decomposition, execution feedback, and validation that free-form text alone does not provide.","primary_link":"https://aclanthology.org/2025.emnlp-main.130/","links":[],"link_count":2,"sections":9},{"id":"codecontests-o-feedback-driven-test-generation-2026","title":"CodeContests-O: Powering LLMs via Feedback-Driven Iterative Test Case Generation","year":2025,"venue":"arXiv","authors":["Jianfeng Cai","Jinhua Zhu","Ruopei Sun","Kangwen Zhao","Dongyun Xue","Mingxiao Feng","Wengang Zhou","Houqiang Li"],"authors_zh":"Jianfeng Cai, Jinhua Zhu, Ruopei Sun, Kangwen Zhao, Dongyun Xue, Mingxiao Feng, Wengang Zhou, Houqiang Li","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["competitive-programming","code-generation"],"tags":["code","test-generation","competitive-programming","rlvr","2026"],"status":"verified","priority":"可读","paper_type_zh":"反馈驱动的竞赛编程测试数据集","best_for_zh":"需要高区分度单测、代码 RLVR 奖励或竞赛编程验证数据的研究者。","confidence":"high","one_line":["CodeContests-O iteratively improves competitive-programming test cases using execution feedback from correct and incorrect solution pools.","CodeContests-O 利用正确与错误解池的执行反馈迭代改进竞赛编程测试用例，从而增强代码结果核验。"],"why":"It makes test quality an iteratively audited executable signal rather than relying on static, sparse cases.","primary_link":"https://arxiv.org/abs/2601.13682","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/cai-jianfeng/CodeContests-O"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/caijanfeng/CodeContests-O"}],"link_count":5,"sections":9},{"id":"codecriticbench-holistic-code-critique-2025","title":"CodeCriticBench: A Holistic Code Critique Benchmark for Large Language Models","year":2025,"venue":"arXiv 2025","authors":["Alexander Zhang","Marcus Dong","Jiaheng Liu","Wei Zhang","Yejie Wang","Jian Yang","Ge Zhang","Tianyu Liu","Zhongyuan Peng","Yingshui Tan","Yuanxing Zhang","Zhexu Wang","Weixun Wang","Yancheng He","Ken Deng","Wangchunshu Zhou","Wenhao Huang","Zhaoxiang Zhang"],"authors_zh":"Alexander Zhang, Marcus Dong, Jiaheng Liu, Wei Zhang, Yejie Wang, Jian Yang, Ge Zhang, Tianyu Liu, Zhongyuan Peng, Yingshui Tan, Yuanxing Zhang, Zhexu Wang, Weixun Wang, Yancheng He, Ken Deng, Wangchunshu Zhou, Wenhao Huang, Zhaoxiang Zhang","tracks":["preference_reward_feedback_data","judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["code-reasoning","critique","evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"代码批评能力评测与细粒度反馈基准论文","best_for_zh":"需要评估或训练代码生成、代码问答场景下的错误定位、解释与修正反馈器的研究者。","confidence":"high","one_line":["CodeCriticBench evaluates code critics with code-generation and code-QA records, execution checks, and fine-grained critique checklists.","万级代码生成与问答批评记录，结合测试与人工维度标签，可训练代码推理的批评和修正反馈器。"],"why":"It exposes the concrete feedback dimensions needed to assess code-reasoning critique rather than relying solely on final code correctness.","primary_link":"https://arxiv.org/abs/2502.16614","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/multimodal-art-projection/CodeCriticBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/m-a-p/CodeCriticBench"},{"key":"project","label":["Project","项目主页"],"url":"https://xxzcc.github.io/CodeCriticBench.github.io/"}],"link_count":5,"sections":9},{"id":"codeelo-2025","title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","year":2025,"venue":"arXiv preprint","authors":["Shanghaoran Quan","Jiaxi Yang","Bowen Yu","Bo Zheng","Dayiheng Liu","An Yang","Xuancheng Ren","Bofei Gao","Yibo Miao","Yunlong Feng","Zekun Wang","Jian Yang","Zeyu Cui","Yang Fan","Yichang Zhang","Binyuan Hui","Junyang Lin"],"authors_zh":"Shanghaoran Quan 等（Qwen Team, Alibaba Group）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit","reward_verifier_layer"],"domains":["competitive-programming","live-hidden-contamination-audit"],"tags":["benchmark","live_hidden_contamination_audit","competitive-programming"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["CodeElo exposes recent Codeforces problems submitted to official judge as an auditable evaluation surface.","CodeElo 把提交至官方评测机的近期 Codeforces 题目做成可审计的评测面。"],"why":"Official platform judging plus recent tasks are strong leakage controls.","primary_link":"https://arxiv.org/abs/2501.01257","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/CodeElo"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Qwen/CodeElo"},{"key":"project","label":["Project","项目主页"],"url":"https://codeelo-bench.github.io/"}],"link_count":5,"sections":9},{"id":"codeultrafeedback-coding-preferences-2025","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","year":2025,"venue":"ACM Transactions on Software Engineering and Methodology","authors":["Martin Weyssow","Aton Kamanda","Xin Zhou","Houari Sahraoui"],"authors_zh":"Martin Weyssow、Aton Kamanda、Xin Zhou、Houari Sahraoui","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences releases 10,000 coding instructions, four candidate answers from fourteen models, and GPT-3.5 scores plus textual feedback across five coding preferences for preference learning, reward modeling, evaluation, and feedback audit.","为编程指令的多模型回答提供五维裁判评分和文字反馈，可训练代码偏好与奖励模型。"],"why":"为编程指令的多模型回答提供五维裁判评分和文字反馈，可训练代码偏好与奖励模型。","primary_link":"https://arxiv.org/abs/2403.09032","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/coseal/CodeUltraFeedback"}],"link_count":2,"sections":9},{"id":"cogcom-2025","title":"CogCoM: A Visual Language Model with Chain-of-Manipulations Reasoning","year":2025,"venue":"ICLR 2025","authors":["Ji Qi","Ming Ding","Weihan Wang","Yushi Bai","Qingsong Lv","Wenyi Hong","Bin Xu","Lei Hou","Juanzi Li","Yuxiao Dong","Jie Tang"],"authors_zh":"Ji Qi, Ming Ding, Weihan Wang, Yushi Bai, Qingsong Lv, Wenyi Hong, Bin Xu, Lei Hou, Juanzi Li, Yuxiao Dong, Jie Tang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","visual-question-answering","visual-grounding","mathematics","ocr"],"tags":["visual-manipulation-traces","multimodal-rationale-data","visual-grounding","ocr","expert-math-data"],"status":"verified","priority":"必读","paper_type_zh":"视觉操作链数据集与多模态模型报告","best_for_zh":"适合构建或审计视觉定位、OCR、计算与图形推理示范数据的读者。","confidence":"high","one_line":["CogCoM releases visual manipulation chains whose grounding, OCR, and calculation results form evidence-bearing multimodal SFT targets.","CogCoM 公开由 GPT-4 规划、视觉工具执行并以金答案终止筛选的视觉操作链，用作带证据的多模态 SFT 目标。"],"why":"It exposes a structured bridge from visual tools and golden answers to trainable multi-turn reasoning demonstrations.","primary_link":"https://arxiv.org/abs/2402.04236","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zai-org/CogCoM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/qijimrc/CoMDataset"}],"link_count":4,"sections":9},{"id":"collaborative-reasoner-coral-2025","title":"Collaborative Reasoner: Self-Improving Social Agents with Synthetic Conversations","year":2025,"venue":"NeurIPS 2025 Main Conference Track","authors":["Ansong Ni","Ruta Desai","Yang Li","Xinjie Lei","Dong Wang","Jiemin Zhang","Jane Yu","Ramya Raghavendra","Gargi Ghosh","Shang-Wen Li","Asli Celikyilmaz"],"authors_zh":"Ansong Ni、Ruta Desai、Yang Li、Xinjie Lei、Dong Wang、Jiemin Zhang、Jane Yu、Ramya Raghavendra、Gargi Ghosh、Shang-Wen Li、Asli Celikyilmaz","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","model_report","benchmark"],"verification_contract":["mixed","judgment_required","programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["collaborative_reasoning","multi_agent_reasoning","mathematical_reasoning","code_reasoning","scientific_reasoning","social_reasoning"],"tags":["collaborative-reasoner","coral","multi-agent-reasoning","synthetic-conversations","self-play","tree-sampling","belief-extraction","dpo","code-only-release"],"status":"partial","priority":"必读","paper_type_zh":"合成多轮协作推理对话的构造配方、混合验证器与 DPO/SFT 模型报告","best_for_zh":"研究多智能体自博弈、对话树采样、基于信念的反馈、同前缀偏好构造，或审计仅开源代码这类发布的读者","confidence":"medium","one_line":["Collaborative Reasoner uses symmetric same-model agents, five-way turn sampling, LLM belief extraction, and gold-answer matching to create next-turn SFT targets and DPO pairs for collaborative reasoning.","Collaborative Reasoner（Coral）让两个对称的同模型 agents 对 reasoning problems 进行五路逐 turn 采样与五棵独立 conversation trees，通过 same-family LLM belief extraction 加 gold matching 构造 SFT targets 和 same-prefix DPO pairs；论文报告 8B/70B 接受 379.6K/311.3K turns，但没有公开 conversations、SFT/DPO rows、rejects、checkpoints、logs 或 splits。"],"why":"It makes the conversational object and feedback contract explicit, while binary answer-only labels, absent rejects/data/checkpoints, and public-config gaps show why open code is not an auditable data release.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/file/221ae0f5de12f7b9803af2656ee7902d-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/collaborative-reasoner"},{"key":"project","label":["Project","项目主页"],"url":"https://ai.meta.com/research/publications/collaborative-reasoner-self-improving-social-agents-with-synthetic-conversations/"}],"link_count":9,"sections":9},{"id":"jpo-joint-instruction-response-preference-2025","title":"Comparing Bad Apples to Good Oranges: Aligning Large Language Models via Joint Preference Optimization","year":2025,"venue":"Findings of ACL 2025","authors":["Hritik Bansal","Ashima Suvarna","Gantavya Bhatt","Nanyun Peng","Kai-Wei Chang","Aditya Grover"],"authors_zh":"Hritik Bansal, Ashima Suvarna, Gantavya Bhatt, Nanyun Peng, Kai-Wei Chang, Aditya Grover（加州大学洛杉矶分校、华盛顿大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["joint-preferences","preference-acquisition","dpo","alignment"],"status":"verified","priority":"必读","paper_type_zh":"联合偏好采集与优化目标研究","best_for_zh":"适合设计需要捕捉非同一指令间相对帮助性或任务完成度的偏好数据集的读者。","confidence":"high","one_line":["JPO learns from comparisons between complete instruction-response pairs, expanding preference optimization beyond fixed-context response rankings.","JPO 从完整的指令—回答对之间学习比较，使偏好优化超出固定上下文下的回答排序。"],"why":"It exposes preference signal that ordinary within-instruction rankings cannot record and makes that signal directly consumable by an alignment objective.","primary_link":"https://aclanthology.org/2025.findings-acl.39/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Hritikbansal/jpo"}],"link_count":3,"sections":9},{"id":"compassverifier-2025","title":"CompassVerifier: A Unified and Robust Verifier for LLMs Evaluation and Outcome Reward","year":2025,"venue":"EMNLP 2025","authors":["Shudong Liu","Hongwei Liu","Junnan Liu","Linchen Xiao","Songyang Gao","Chengqi Lyu","Yuzhe Gu","Wenwei Zhang","Derek F. Wong","Songyang Zhang","Kai Chen"],"authors_zh":"Shudong Liu, Hongwei Liu, Junnan Liu, Linchen Xiao, Songyang Gao, Chengqi Lyu, Yuzhe Gu, Wenwei Zhang, Derek F. Wong, Songyang Zhang, Kai Chen","tracks":["data_construction_open_release_recipes","preference_reward_feedback_data"],"source_role":["benchmark","verifier_reward","data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["reward_modeling","rlvr","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics","general_reasoning","knowledge","science"],"tags":["outcome-verifier","verifierbench","reward-model","human-adjudication","meta-error-patterns","synthetic-augmentation","multi-model-voting","reasoning-evaluation","open-release","release-gap"],"status":"partial","priority":"必读","paper_type_zh":"Outcome verifier、评测基准与开放发布","best_for_zh":"构建答案过滤器、基于参考答案的奖励、Verifier benchmark，或审计数学、科学、知识与通用推理 RLVR 流水线的研究者","confidence":"high","one_line":["A staged outcome-verifier pipeline with ternary human labels, error-driven synthesis, a public benchmark, and three public Qwen2.5-based checkpoints.","该工作将 53 个模型的回答池加工为 2,817 条三分类 VerifierBench，并发布三个 CompassVerifier 检查点；但 96,832 条训练语料、投票与过滤谱系以及完整构建代码仍未发布。"],"why":"It makes answer verification itself a trainable and testable data object for evaluation and RL rewards, while exposing label-policy, teacher concentration, binary-collapse, and release-completeness risks.","primary_link":"https://aclanthology.org/2025.emnlp-main.1698/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/open-compass/CompassVerifier"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/opencompass/VerifierBench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/opencompass/compassverifier"},{"key":"project","label":["Project","项目主页"],"url":"https://open-compass.github.io/CompassVerifier/"}],"link_count":7,"sections":9},{"id":"cvdp-verilog-design-verification-2025","title":"Comprehensive Verilog Design Problems: A Next-Generation Benchmark Dataset for Evaluating Large Language Models and Agents on RTL Design and Verification","year":2025,"venue":"arXiv preprint","authors":["Nathaniel Pinckney","Chenhui Deng","Chia-Tung Ho","Yun-Da Tsai","Mingjie Liu","Wenfei Zhou","Brucek Khailany","Haoxing Ren"],"authors_zh":"Nathaniel Pinckney、Chenhui Deng、Chia-Tung Ho、Yun-Da Tsai、Mingjie Liu、Wenfei Zhou、Brucek Khailany、Haoxing Ren","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["verilog","hardware-design","agents"],"tags":["verilog","rtl","hardware-agents","2025"],"status":"verified","priority":"可读","paper_type_zh":"Verilog RTL 设计与验证基准","best_for_zh":"研究硬件设计代理、RTL 生成、验证和调试的研究者。","confidence":"high","one_line":["CVDP releases 783 engineer-authored RTL generation, verification, and debugging problems with tool-backed evaluation for models and agents.","CVDP 发布覆盖 RTL 生成、验证与调试的 783 个工程师编写问题，以工具执行和评分基础设施评测模型与代理。"],"why":"CVDP releases 783 engineer-authored RTL generation, verification, and debugging problems with tool-backed evaluation for models and agents.","primary_link":"https://arxiv.org/abs/2506.14074","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVlabs/cvdp_benchmark"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/cvdp-benchmark-dataset"}],"link_count":3,"sections":9},{"id":"covo-self-rewarding-2025","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Kongcheng Zhang","Qi Yao","Shunyu Liu","Yingjie Wang","Baisheng Lai","Jieping Ye","Mingli Song","Dacheng Tao"],"authors_zh":"Kongcheng Zhang、Qi Yao、Shunyu Liu、Yingjie Wang、Baisheng Lai、Jieping Ye、Mingli Song、Dacheng Tao","tracks":["data_construction_open_release_recipes","preference_reward_feedback_data"],"source_role":["data_release","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward","trajectory_value"],"training_use":["unknown","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematical_reasoning","commonsense_reasoning","science_reasoning"],"tags":["self-rewarding-rl","trajectory-consistency","volatility","curiosity-reward","reinforce-plus-plus","reward-hacking"],"status":"partial","priority":"必读","paper_type_zh":"自奖励推理强化学习配方","best_for_zh":"关注内在奖励、轨迹评价和奖励攻击审计的研究者","confidence":"medium","one_line":["Converts 16 on-policy trajectories into consistency, volatility, answer-group, and curiosity rewards for Reinforce++ without using correctness labels for optimization.","把 16 条策略轨迹转换为一致性、波动性和好奇心奖励，用于无需外部正确性标签的 Reinforce++。"],"why":"Makes a self-generated reward recipe inspectable while showing why internal consistency must not be conflated with external correctness.","primary_link":"https://openreview.net/forum?id=ckW70ls93V","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sastpg/CoVo"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sastpg/CoVo_Dataset"}],"link_count":6,"sections":9},{"id":"copr-continual-preference-2025","title":"COPR: Continual Human Preference Learning via Optimal Policy Regularization","year":2025,"venue":"Findings of ACL 2025","authors":["Han Zhang","Lin Gui","Yu Lei","Yuanzhao Zhai","Yehong Zhang","Zhuo Zhang","Yulan He","Hui Wang","Yue Yu","Kam-Fai Wong","Bin Liang","Ruifeng Xu"],"authors_zh":"Han Zhang 等（哈尔滨工业大学、鹏城实验室、伦敦国王学院等）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","continual_learning","preference_learning"],"tags":["continual-alignment","preference-learning","policy-regularization","data-sequences"],"status":"verified","priority":"可读","paper_type_zh":"持续偏好优化与策略正则化研究","best_for_zh":"适合构建持续接收偏好数据的对齐系统，并需在适应新任务时保留既有行为的读者。","confidence":"high","one_line":["COPR constrains continual preference updates with historical optimal-policy distributions so new feedback can be learned without collapsing earlier preferences.","COPR 用历史最优策略分布约束持续偏好更新，使模型学习新反馈时不至于遗忘或压扁早期偏好。"],"why":"It makes the arrival order and historical consumption of preference data explicit components of the training objective.","primary_link":"https://aclanthology.org/2025.findings-acl.281/","links":[],"link_count":2,"sections":9},{"id":"cot-valve-2025","title":"CoT-Valve: Length-Compressible Chain-of-Thought Tuning","year":2025,"venue":"ACL 2025","authors":["Xinyin Ma","Guangnian Wan","Runpeng Yu","Gongfan Fang","Xinchao Wang"],"authors_zh":"Xinyin Ma；Guangnian Wan；Runpeng Yu；Gongfan Fang；Xinchao Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","process_reward"],"training_use":["sft","distillation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["chain-of-thought","compression","length-control","mixchain","process-reward-model"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A released mixed-length CoT resource and tuning recipe for controllable reasoning compression.","CoT-Valve 使用 MixChain 中不同长度的推理链训练长度可压缩的 CoT，并把压缩与正确性筛选作为可审计的构造步骤。"],"why":"It highlights that a short trace is not automatically auditable: filtering flags, source lineage, and compression decisions must accompany the text.","primary_link":"https://aclanthology.org/2025.acl-long.300/","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/horseee/cot-valve"}],"link_count":3,"sections":9},{"id":"critic-cot-chain-of-thought-critic-2025","title":"Critic-CoT: Boosting the Reasoning Abilities of Large Language Model via Chain-of-Thought Critic","year":2025,"venue":"Findings of ACL 2025","authors":["Xin Zheng","Jie Lou","Boxi Cao","Xueru Wen","Yuqiu Ji","Hongyu Lin","Yaojie Lu","Xianpei Han","Debing Zhang","Le Sun"],"authors_zh":"Xin Zheng, Jie Lou, Boxi Cao, Xueru Wen, Yuqiu Ji, Hongyu Lin, Yaojie Lu, Xianpei Han, Debing Zhang, Le Sun","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","self-critique","weak-supervision"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"逐步思维链批评与弱监督反馈数据论文","best_for_zh":"需要以逐步批评信号筛除错误数学解答，或迭代修复模型推理轨迹的研究者。","confidence":"high","one_line":["Critic-CoT releases weakly supervised step-wise critiques and revisions so models can reject or repair flawed mathematical reasoning chains.","27.6万数学批评／改写实例，弱监督产生逐步批评反馈，可用于筛除和迭代修复错误推理。"],"why":"It treats critique as a structured feedback process that can be used both to select solutions and to refine them.","primary_link":"https://aclanthology.org/2025.findings-acl.89/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ExpectoZX/critic_cot"}],"link_count":5,"sections":9},{"id":"critic-v-vlm-critics-multimodal-reasoning-2025","title":"Critic-V: VLM Critics Help Catch VLM Errors in Multimodal Reasoning","year":2025,"venue":"CVPR 2025","authors":["Di Zhang","Jingdi Lei","Junxian Li","Xunzhi Wang","Yujie Liu","Zonglin Yang","Jiatong Li","Weida Wang","Suorong Yang","Jianbo Wu","Peng Ye","Wanli Ouyang","Dongzhan Zhou"],"authors_zh":"Di Zhang、Jingdi Lei、Junxian Li、Xunzhi Wang、Yujie Liu、Zonglin Yang、Jiatong Li、Weida Wang、Suorong Yang、Jianbo Wu、Peng Ye、Wanli Ouyang、Dongzhan Zhou","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","full_episode"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","vision-language"],"tags":["vision-language","critic","dpo","cvpr-2025"],"status":"verified","priority":"可读","paper_type_zh":"多模态 critic 偏好数据与 DPO 方法","best_for_zh":"需要训练视觉语言 critic、自然语言反馈或多模态推理纠错模型的研究者","confidence":"high","one_line":["Critic-V releases 29,012 multimodal critique preference pairs and trains a DPO critic that gives natural-language feedback to refine VLM reasoning.","发布按规则奖励排序的多模态批评偏好数据，用自然语言 critique 修正视觉推理路径，可直接训练 critic 与策略模型。"],"why":"It replaces a scalar reward with preference-optimized textual critiques that can identify and guide correction of a reasoning error.","primary_link":"https://openaccess.thecvf.com/content/CVPR2025/html/Zhang_Critic-V_VLM_Critics_Help_Catch_VLM_Errors_in_Multimodal_Reasoning_CVPR_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kyrieLei/Critic-V"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/huaXiaKyrie/critique-VQA"}],"link_count":4,"sections":9},{"id":"critique-fine-tuning-cft-2025","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","year":2025,"venue":"COLM 2025","authors":["Yubo Wang","Xiang Yue","Wenhu Chen"],"authors_zh":"Yubo Wang, Xiang Yue, Wenhu Chen","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","stem-reasoning","critique"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"批评微调与噪声回答反馈数据论文","best_for_zh":"需要从错误或可疑回答中训练模型诊断、解释和修复推理问题的数学与 STEM 研究者。","confidence":"high","one_line":["WebInstruct-CFT teaches models to diagnose noisy solutions with GPT-4o-generated critiques instead of merely imitating correct answers.","65.4万问题—噪声解—批评记录（论文用50k子集），以文字批评促进数学与 STEM 推理修正。"],"why":"The release turns textual error analysis into a reusable training object rather than reducing feedback to a correct-answer target.","primary_link":"https://arxiv.org/abs/2501.17703","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TIGER-AI-Lab/CritiqueFineTuning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TIGER-Lab/WebInstruct-CFT"},{"key":"project","label":["Project","项目主页"],"url":"https://tiger-ai-lab.github.io/CritiqueFineTuning/"}],"link_count":5,"sections":9},{"id":"cross-lingual-auto-evaluation-2025","title":"Cross-Lingual Auto Evaluation for Assessing Multilingual LLMs","year":2025,"venue":"ACL 2025","authors":["Sumanth Doddapaneni","Mohammed Safi Ur Rahman Khan","Dilip Venkatesh","Raj Dabre","Anoop Kunchukuttan","Mitesh M. Khapra"],"authors_zh":"Sumanth Doddapaneni、Mohammed Safi Ur Rahman Khan、Dilip Venkatesh、Raj Dabre、Anoop Kunchukuttan、Mitesh M. Khapra","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["The CIA Suite's RECON provides human-scored instructions in six languages for training and meta-evaluating cross-lingual judges.","CIA Suite 的 RECON 以六语种人工判分指令数据，支持跨语言自动评测器的训练与元评测。"],"why":"The CIA Suite's RECON provides human-scored instructions in six languages for training and meta-evaluating cross-lingual judges.","primary_link":"https://aclanthology.org/2025.acl-long.1419/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ai4bharat/recon"}],"link_count":2,"sections":9},{"id":"crossing-reward-bridge-rlvr-2025","title":"Crossing the Reward Bridge: Expanding Reinforcement Learning with Verifiable Rewards Across Diverse Domains","year":2025,"venue":"ACL 2026","authors":["Yi Su","Dian Yu","Linfeng Song","Juntao Li","Haitao Mi","Zhaopeng Tu","Min Zhang","Dong Yu"],"authors_zh":"Yi Su、Dian Yu、Linfeng Song、Juntao Li、Haitao Mi、Zhaopeng Tu、Min Zhang、Dong Yu","tracks":["training_usage_optimization_objectives"],"source_role":["verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["scalar_reward"],"training_use":["rlvr"],"construction_layer":["reward_verifier_layer"],"domains":["multi-domain-reasoning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"跨领域 RLVR 与生成式奖励模型论文","best_for_zh":"希望把自由文本参考答案转成可训练 RL 奖励的读者。","confidence":"high","one_line":["This work distils cross-domain generative reward models from exploration-time answer–judgement records and uses soft rewards for RLVR.","该工作用探索中收集的回答—判定记录蒸馏跨领域生成式奖励模型，并以软奖励扩展自由文本任务的 RLVR。"],"why":"It makes the link between an explicit data object and a training objective inspectable.","primary_link":"https://aclanthology.org/2026.acl-long.178/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/virtuoussy/rlvr"}],"link_count":4,"sections":9},{"id":"crpo-confidence-reward-preference-2025","title":"CRPO: Confidence-Reward Driven Preference Optimization for Machine Translation","year":2025,"venue":"Findings of ACL 2025","authors":["Guofeng Cui","Pichao Wang","Yang Liu","Zemian Ke","Zhu Liu","Vimal Bhat"],"authors_zh":"Guofeng Cui, Pichao Wang, Yang Liu, Zemian Ke, Zhu Liu, Vimal Bhat","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["machine_translation","preference_learning"],"tags":["preference-data","data-selection","dpo","translation"],"status":"verified","priority":"可读","paper_type_zh":"置信度与奖励联合的偏好数据选择研究","best_for_zh":"适合构造翻译偏好对，并需要区分高质量样本与当前模型仍能从中学习的样本的读者。","confidence":"high","one_line":["CRPO selects translation preference pairs with both reward contrast and model uncertainty, concentrating DPO on records with higher learning value.","CRPO 同时依据奖励差距与模型置信度选择翻译偏好对，让 DPO 聚焦仍有学习价值的记录。"],"why":"It makes model confidence a selection signal alongside reward, so preference data is chosen for expected learning value rather than quality alone.","primary_link":"https://aclanthology.org/2025.findings-acl.31/","links":[],"link_count":2,"sections":9},{"id":"cuarewardbench-computer-using-agent-reward-models-2025","title":"CUARewardBench: A Benchmark for Evaluating Reward Models on Computer-using Agent","year":2025,"venue":"arXiv preprint","authors":["Haojia Lin","Xiaoyu Tan","Yulei Qin","Zihan Xu","Yuchen Shi","Zongyi Li","Gang Li","Shaofei Cai","Siqi Cai","Chaoyou Fu","Ke Li","Xing Sun"],"authors_zh":"Haojia Lin、Xiaoyu Tan、Yulei Qin、Zihan Xu、Yuchen Shi、Zongyi Li、Gang Li、Shaofei Cai、Siqi Cai、Chaoyou Fu、Ke Li、Xing Sun","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["computer-using-agents","reward-model-evaluation","process-supervision"],"tags":["process-supervision","agent-trajectories","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要比较计算机使用智能体奖励模型、核查局部操作错误或评估轨迹级反馈质量的研究者。","confidence":"high","one_line":["CUARewardBench evaluates whether reward models can judge computer-use trajectories at both the step and whole-trajectory levels across diverse software tasks.","CUARewardBench 在多类软件任务上同时评测奖励模型对计算机操作轨迹的逐步判断和整轨判断能力。"],"why":"It supplies a public evaluation surface for reward models in computer-use settings, where process mistakes can compound through an action trajectory.","primary_link":"https://arxiv.org/abs/2510.18596","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent/CUARewardBench"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Tencent/CUARewardBench/tree/main/data/annotations"}],"link_count":4,"sections":9},{"id":"dapo-2025","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","year":2025,"venue":"arXiv","authors":["Qiying Yu","Zheng Zhang","Ruofei Zhu","Yufeng Yuan","Xiaochen Zuo","Yu Yue","Weinan Dai","Tiantian Fan","Gaohong Liu","Lingjun Liu","Xin Liu","Haibin Lin","Zhiqi Lin","Bole Ma","Guangming Sheng","Yuxuan Tong","Chi Zhang","Mofan Zhang","Wang Zhang","Hang Zhu","Jinhua Zhu","Jiaze Chen","Jiangjie Chen","Chengyi Wang","Hongli Yu","Yuxuan Song","Xiangpeng Wei","Hao Zhou","Jingjing Liu","Wei-Ying Ma","Ya-Qin Zhang","Lin Yan","Mu Qiao","Yonghui Wu","Mingxuan Wang"],"authors_zh":"Qiying Yu 等","tracks":["data_construction_open_release_recipes","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study","infrastructure"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","optimizer_scaffold","reward_verifier_layer","scaling_report","release_audit"],"domains":["math","rlvr","reasoning"],"tags":["curated-card","primary-link-checked","artifact-verified","rlvr","math","open-release","dynamic-sampling","rule-reward"],"status":"verified","priority":"必读","paper_type_zh":"开源数学 RLVR 算法、系统与数据构造配方","best_for_zh":"设计、复现或审计数学 RLVR 数据流与反馈合同的研究者","confidence":"high","one_line":["DAPO couples a 17K integer-answer math prompt recipe with rule-scored online rollouts, dynamic group filtering, token-level loss, and length shaping, but the current public split and source lineage require audit before reuse.","DAPO 以 17K 整数答案数学题和每题 16 条在线 rollout 为对象，通过规则奖励、动态题组筛选、token-level loss 与长度塑形组织 RLVR；但当前公开 split 的重复块和来源谱系仍阻碍直接复用。"],"why":"For the Data Construction and Open Release track, DAPO shows that the effective RLVR dataset is not only the prompt file: answer transformation, verifier behavior, rollout multiplicity, prompt-group filtering, sequence-length shaping, and release version jointly decide which tokens receive gradient.","primary_link":"https://arxiv.org/abs/2503.14476","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/BytedTsinghua-SIA/DAPO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/BytedTsinghua-SIA/DAPO-Math-17k"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/BytedTsinghua-SIA/dapo"},{"key":"project","label":["Project","项目主页"],"url":"https://dapo-sia.github.io/"}],"link_count":8,"sections":9},{"id":"dast-difficulty-adaptive-slow-thinking-2025","title":"DAST: Difficulty-Adaptive Slow-Thinking for Large Reasoning Models","year":2025,"venue":"EMNLP 2025 Industry Track","authors":["Yi Shen","Jian Zhang","Jieyun Huang","Shuming Shi","Wenjing Zhang","Jiangze Yan","Ning Wang","Kai Wang","Zhaoxiang Liu","Shiguo Lian"],"authors_zh":"Yi Shen、Jian Zhang、Jieyun Huang、Shuming Shi、Wenjing Zhang、Jiangze Yan、Ning Wang、Kai Wang、Zhaoxiang Liu、Shiguo Lian","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["preference_learning","test_time_compute"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics"],"tags":["reasoning-efficiency","adaptive-compute","self-rollouts","token-budget","preference-pairs","simpo","test-time-compute","reward-shaping"],"status":"partial","priority":"必读","paper_type_zh":"难度自适应推理偏好数据构造与 SimPO 训练配方","best_for_zh":"关注 self-rollout、长度预算、偏好对构造与测试时推理计算分配的读者","confidence":"medium","one_line":["DAST samples 20 self-rollouts per MATH question, uses correctness and a difficulty-conditioned token budget to rank candidates, and trains SimPO on the resulting preference pairs.","DAST 对每个 MATH 训练题收集 20 条自生成推理响应，以正确性和难度相关的 Token Length Budget 校准奖励、构造偏好对，并用 SimPO 训练长度自适应的慢思考行为。"],"why":"It makes the rollout budget, token-budget construction, reward-shaped pair ranking, and length-sensitive preference optimization visible, while leaving the reusable records and checker implementation unavailable.","primary_link":"https://aclanthology.org/2025.emnlp-industry.160/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AnonymousUser0520/AnonymousRepo01"}],"link_count":4,"sections":9},{"id":"data-contamination-quiz-2025","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","year":2025,"venue":"TACL 2025","authors":["Shahriar Golchin","Mihai Surdeanu"],"authors_zh":"Shahriar Golchin, Mihai Surdeanu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"黑盒数据污染检测与估计方法","best_for_zh":"需要在无法取得训练数据、权重或 logits 时审计基准污染的研究者。","confidence":"high","one_line":["Accepted TACL paper; official repository releases the quiz inputs, code, and contamination reports.","以原题与词级扰动组成选择题，在黑盒模型中估计逐字数据污染及其位置偏差边界。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.tacl-1.37/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/shahriargolchin/DCQ"}],"link_count":3,"sections":9},{"id":"data-mixing-optimization-sft-2025","title":"Data Mixing Optimization for Supervised Fine-Tuning of Large Language Models","year":2025,"venue":"ICML 2025","authors":["Yuan Li","Zhengzhong Liu","Eric Xing"],"authors_zh":"Yuan Li, Zhengzhong Liu, Eric Xing","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["instruction-tuning","mathematics","code"],"tags":["data-mixing","supervised-fine-tuning","validation-loss","scaling-laws"],"status":"verified","priority":"必读","paper_type_zh":"监督微调数据混合优化研究","best_for_zh":"为昂贵的多领域监督微调选择领域比例的读者。","confidence":"high","one_line":["A fitted transfer-and-scaling loss model selects SFT domain weights for a fixed token budget.","拟合迁移—缩放损失模型，在固定词元预算下选择监督微调的领域权重。"],"why":"It turns data-source allocation into an explicit, auditable optimization decision.","primary_link":"https://proceedings.mlr.press/v267/li25bh.html","links":[],"link_count":2,"sections":9},{"id":"data-whisperer-task-selection-2025","title":"Data Whisperer: Efficient Data Selection for Task-Specific LLM Fine-Tuning via Few-Shot In-Context Learning","year":2025,"venue":"ACL 2025","authors":["Shaobo Wang","Xiangqi Jin","Ziming Wang","Jize Wang","Jiajun Zhang","Kaixin Li","Zichen Wen","Zhong Li","Conghui He","Xuming Hu","Linfeng Zhang"],"authors_zh":"Shaobo Wang, Xiangqi Jin, Ziming Wang, Jize Wang, Jiajun Zhang, Kaixin Li, Zichen Wen, Zhong Li, Conghui He, Xuming Hu, Linfeng Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["task_specific_finetuning","reasoning","instruction_tuning"],"tags":["data-selection","in-context-learning","attention","task-specific-finetuning","efficient-training"],"status":"verified","priority":"必读","paper_type_zh":"免训练的任务数据选择研究","best_for_zh":"适合能够查看模型注意力、却无力承担目标语料评分微调成本的任务特定监督微调子集选择者。","confidence":"high","one_line":["Data Whisperer selects task-specific fine-tuning records by measuring how attention-weighted few-shot demonstrations improve query performance before any scoring-model training.","Data Whisperer 在任何评分模型训练之前，衡量带注意力权重的少样本示例如何改善查询表现，并据此选择任务微调记录。"],"why":"It makes pre-training ICL behavior an explicit training-record utility signal and evaluates selection cost relative to the full fine-tuning budget.","primary_link":"https://aclanthology.org/2025.acl-long.1135/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/gszfwsb/Data-Whisperer"}],"link_count":3,"sections":9},{"id":"datascibench-2025","title":"DataSciBench: An LLM Agent Benchmark for Data Science","year":2025,"venue":"arXiv preprint","authors":["Dan Zhang","Sining Zhoubian","Min Cai","Fengzu Li","Lekang Yang","Wei Wang","Tianjiao Dong","Ziniu Hu","Jie Tang","Yisong Yue"],"authors_zh":"Dan Zhang 等（Tsinghua University、Zhipu AI、University of California, Berkeley、California Institute of Technology）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["data-science-agents","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","data-science-agents"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"low","one_line":["DataSciBench exposes data-science tasks with executable metric framework as an auditable evaluation surface.","DataSciBench 把配有可执行指标框架的数据科学任务做成可审计的评测面。"],"why":"Likely relevant, but exact publication/source needs confirmation.","primary_link":"https://arxiv.org/abs/2502.13897","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zd21/DataSciBench"},{"key":"project","label":["Project","项目主页"],"url":"https://datascibench.github.io/"}],"link_count":5,"sections":9},{"id":"dcrm-pair-quality-2025","title":"DCRM: A Heuristic to Measure Response Pair Quality in Preference Optimization","year":2025,"venue":"Findings of EMNLP 2025","authors":["Chengyu Huang","Tanya Goyal"],"authors_zh":"Chengyu Huang, Tanya Goyal（康奈尔大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-data","response-pair-selection","dpo","reward-margin"],"status":"verified","priority":"必读","paper_type_zh":"偏好对质量度量与数据选择研究","best_for_zh":"适合构造离线偏好优化的优劣回答对，并需要超越单纯奖励差筛选标准的读者。","confidence":"high","one_line":["DCRM selects preference pairs whose reward-relevant contrast is large relative to their noisy lexical and model-probability differences.","DCRM 以奖励相关差异相对于词面和模型概率噪声的密度来选择偏好对，避免仅按奖励差距训练。"],"why":"It makes the usefulness of a preference pair a training-data decision, connecting pair construction directly to downstream preference-optimization outcomes.","primary_link":"https://aclanthology.org/2025.findings-emnlp.136/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HCY123902/DCRM"}],"link_count":3,"sections":9},{"id":"pfp-online-preference-debiasing-2025","title":"Debiasing Online Preference Learning via Preference Feature Preservation","year":2025,"venue":"Findings of ACL 2025","authors":["Dongyoung Kim","Jinsung Yoon","Jinwoo Shin","Jaehyung Kim"],"authors_zh":"Dongyoung Kim, Jinsung Yoon, Jinwoo Shin, Jaehyung Kim（KAIST、Google Cloud AI Research、延世大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","online_preference_learning"],"tags":["online-preference-learning","preference-features","curriculum","data-generation"],"status":"verified","priority":"可读","paper_type_zh":"在线偏好数据构造与分布保持优化研究","best_for_zh":"适合构建迭代偏好学习流程，并需要避免自生成反馈累积后特征坍缩的读者。","confidence":"high","one_line":["PFP preserves the feature distribution behind preference labels while generating and training on new online preference records.","PFP 在生成并训练新的在线偏好记录时保留偏好标签背后的特征分布，减少迭代中的特征坍缩。"],"why":"It turns the feature mix behind binary preferences into a constraint on what data is generated and consumed in each online-training round.","primary_link":"https://aclanthology.org/2025.findings-acl.1034/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kingdy2002/PFP"}],"link_count":3,"sections":9},{"id":"deconstructing-long-cot-2025","title":"Deconstructing Long Chain-of-Thought: A Structured Reasoning Optimization Framework for Long CoT Distillation","year":2025,"venue":"arXiv preprint","authors":["Yijia Luo","Yulin Song","Xingyao Zhang","Jiaheng Liu","Weixun Wang","GengRu Chen","Wenbo Su","Bo Zheng"],"authors_zh":"Yijia Luo、Yulin Song、Xingyao Zhang、Jiaheng Liu、Weixun Wang、GengRu Chen、Wenbo Su、Bo Zheng","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["math"],"tags":["long-cot","distillation","trace-segmentation","redundancy-pruning","teacher-traces","math","rule-based-validation"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹、搜索或测试时计算研究","best_for_zh":"需要审计 rollout、选择器、预算、数据谱系与复现边界的读者","confidence":"medium","one_line":["DLCoT parses long mathematical teacher traces into structured approaches and verifications, prunes redundant branches for SFT, and releases processing code plus a Step 1 input rather than a confirmed final optimized corpus.","该条目将推理轨迹、搜索选择或测试时计算作为可审计的研究对象；未披露字段已明确标为 unknown。"],"why":"DLCoT makes trace structure, filtering, clustering, and deletion choices auditable construction variables, but its public repository is a recipe plus Step 1 input rather than a reusable release of the final optimized distillation data.","primary_link":"https://arxiv.org/abs/2503.16385","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/elena-luo/SODE"},{"key":"data","label":["Data","数据"],"url":"https://github.com/elena-luo/SODE/blob/main/data/20250209_deepseek_r1_math_26K.jsonl"}],"link_count":5,"sections":9},{"id":"openai-deep-research-system-card-2025","title":"Deep Research System Card","year":2025,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["rlvr","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","web_browsing","agentic_tool_use","safety"],"tags":["openai","deep-research","o3","system-card","frontier-report","data-disclosure-ledger","web-browsing","reinforcement-learning","prompt-injection","safety"],"status":"partial","priority":"必读","paper_type_zh":"前沿智能体系统卡与浏览后训练数据披露台账","best_for_zh":"需要审计浏览智能体后训练、奖励/评分合约、安全数据与污染边界的读者","confidence":"high","one_line":["OpenAI's Deep Research System Card discloses browsing-task reinforcement learning with ground-truth or rubric grading by a chain-of-thought model and added browsing safety data, but not the datasets, trajectories, graders, reward design, or reproducibility artifacts.","OpenAI 的 Deep Research System Card 披露了浏览任务强化学习：以真值或量规为依据、由思维链模型评分，并加入浏览安全数据；但未披露数据集、轨迹、评分器、奖励设计或复现工件。"],"why":"It is a direct Track 12 disclosure of an agentic browsing post-training interface: it identifies the high-level task and feedback contracts while making the missing provenance, reward, environment, and contamination evidence explicit.","primary_link":"https://cdn.openai.com/deep-research-system-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-deep-research/"}],"link_count":3,"sections":9},{"id":"deepdistill-2025","title":"DeepDistill: Enhancing LLM Reasoning Capabilities via Large-Scale Difficulty-Graded Data Training","year":2025,"venue":"arXiv preprint","authors":["Xiaoyu Tian","Sitong Zhao","Haotian Wang","Shuaiting Chen","Yiping Peng","Yunjie Ji","Han Zhao","Xiangang Li"],"authors_zh":"Xiaoyu Tian、Sitong Zhao、Haotian Wang、Shuaiting Chen、Yiping Peng、Yunjie Ji、Han Zhao、Xiangang Li","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline"],"domains":["mathematics","code","science","instruction_following","general_reasoning"],"tags":["difficulty-graded-data","reasoning-distillation","multi-teacher","open-data","data-filtering","mixed-verification"],"status":"partial","priority":"必读","paper_type_zh":"推理数据发布与构造配方","best_for_zh":"研究开放推理数据、教师蒸馏、难度分级、混合验证器与数据审计的读者","confidence":"high","one_line":["Builds about 40M mixed-domain teacher responses for roughly 3.34M sourced queries and grades them with category-specific verification, while the official data card says the public release is only a subset.","围绕约 334 万条有来源的查询生成约 4000 万条多教师响应，并以分类型验证器给出难度信号；但官方数据卡说明公开内容仅是完整数据的子集。"],"why":"Exposes a large trace-and-feedback object for studying difficulty-aware data selection, together with enough caveats to audit verifier dependence and release completeness.","primary_link":"https://arxiv.org/abs/2504.17565","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/a-m-team/AM-DeepSeek-Distilled-40M"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/a-m-team/a-m-models"}],"link_count":4,"sections":9},{"id":"deepdiver-2025","title":"DeepDiver: Adaptive Web-Search Intensity Scaling via Reinforcement Learning","year":2025,"venue":"NeurIPS 2025","authors":["Wenxuan Shi","Haochen Tan","Chuqiao Kuang","Xiaoguang Li","Hanting Chen","Xiaozhe Ren","Yasheng Wang","Lu Hou","Lifeng Shang"],"authors_zh":"Wenxuan Shi、Haochen Tan、Chuqiao Kuang、Xiaoguang Li、Hanting Chen、Xiaozhe Ren、Yasheng Wang、Lu Hou、Lifeng Shang","tracks":["rollout_search_test_time_trace_data"],"source_role":["benchmark","construction_recipe","scaling_study"],"verification_contract":["environmental","judgment_required"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","distillation","agent_training","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["web","information_seeking","reasoning"],"tags":["deepdiver","webpuzzle","open-web-rl","iterative-rag","search-intensity-scaling","grpo","llm-grader"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["DeepDiver trains 7B policies on WebPuzzle with live-web iterative-RAG GRPO, using answer judgments and a conditional search bonus to study how search depth changes with information demand.","DeepDiver 在 WebPuzzle 上用真实网页的迭代式检索增强生成与 GRPO 训练 7B 策略，以答案评判加条件化搜索奖励研究搜索深度如何随信息需求变化。"],"why":"It makes search queries, retrieved evidence, stopping decisions, and episode rewards relevant rollout data for the rollout/search/test-time trace track, while leaving live-web drift and artifact availability as explicit audit limits.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/180d4373aca26bd86bf45fc50d1a709f-Abstract-Conference.html","links":[],"link_count":5,"sections":9},{"id":"deepmath-103k-2025","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","year":2025,"venue":"arXiv preprint (2025)","authors":["He, Zhiwei","Liang, Tian","Xu, Jiahao","Liu, Qiuzhi","Chen, Xingyu","Wang, Yue","Song, Linfeng","Yu, Dian","Liang, Zhenwen","Wang, Wenxuan","Zhang, Zhuosheng","Wang, Rui","Tu, Zhaopeng","Mi, Haitao","Yu, Dong"],"authors_zh":"He, Zhiwei、Liang, Tian、Xu, Jiahao、Liu, Qiuzhi、Chen, Xingyu、Wang, Yue、Song, Linfeng、Yu, Dian、Liang, Zhenwen、Wang, Wenxuan、Zhang, Zhuosheng、Wang, Rui、Tu, Zhaopeng、Mi, Haitao、Yu, Dong","tracks":["instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English competition and advanced mathematics"],"tags":["instruction-demonstration-rationale","arxiv-2504.11456","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"数学推理监督微调与强化学习题目准备","confidence":"high","one_line":["DeepMath-103K retains difficult decontaminated questions, stores a verified final answer, and supplies three independent R1 reasoning traces for each problem.","DeepMath-103K 为 10.3 万道去污染难题各保留三条 R1 解题过程，并把答案、难度和主题写入记录。"],"why":"Many open math corpora contain easy, duplicated, contaminated, or unverifiable problems, obscuring whether training gains come from reasoning quality or leakage.","primary_link":"https://arxiv.org/abs/2504.11456","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zwhe99/DeepMath"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zwhe99/DeepMath-103K"}],"link_count":4,"sections":9},{"id":"deepresearch-bench-agents-2025","title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","year":2025,"venue":"ICLR 2026","authors":["Mingxuan Du","Benfeng Xu","Chiwei Zhu","Xiaorui Wang","Zhendong Mao"],"authors_zh":"Mingxuan Du, Benfeng Xu, Chiwei Zhu, Xiaorui Wang, Zhendong Mao","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["deep_research","web","cross_domain"],"tags":["deep_research","web","rubric"],"status":"verified","priority":"必读","paper_type_zh":"深度研究 Agent 的专家任务、适应性 rubric 与引文评测基准","best_for_zh":"研究开放网页深度研究、长报告 Judge 与引用质量评估的读者。","confidence":"high","one_line":["DeepResearch Bench evaluates PhD-level web research reports with adaptive criteria and citation-quality assessments.","DeepResearch Bench 用博士级研究任务、适应性准则和引文评估，衡量深度研究 Agent 的报告质量与检索能力。"],"why":"It separates report-quality and retrieval-citation evaluation for deep-research agents.","primary_link":"https://arxiv.org/abs/2506.11763","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Ayanami0730/deep_research_bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/muset-ai/DeepResearch-Bench-Dataset"}],"link_count":5,"sections":9},{"id":"deepresearcher-2025","title":"DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments","year":2025,"venue":"EMNLP 2025","authors":["Yuxiang Zheng","Dayuan Fu","Xiangkun Hu","Xiaojie Cai","Lyumanshan Ye","Pengrui Lu","Pengfei Liu"],"authors_zh":"Yuxiang Zheng, Dayuan Fu, Xiangkun Hu, Xiaojie Cai, Lyumanshan Ye, Pengrui Lu, Pengfei Liu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["agent_training","test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","raw_search_rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["DeepResearcher releases 80,000 filtered QA prompts, code, and a 7B checkpoint for live-web GRPO with 16 rollouts per prompt and short-answer F1 reward, but not a verified corpus of the original training episodes.","DeepResearcher 在真实网页搜索与浏览环境中进行 GRPO，发布处理后的输入、代码和模型但未发布训练轨迹。"],"why":"It makes live search, page reading, tool-observation masking, rollout allocation, and terminal reward explicit, while exposing how dynamic web state limits trajectory replay and audit.","primary_link":"https://aclanthology.org/2025.emnlp-main.22/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/DeepResearcher"},{"key":"data","label":["Data","数据"],"url":"https://github.com/GAIR-NLP/DeepResearcher/tree/main/data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/GAIR/DeepResearcher-7b"}],"link_count":7,"sections":9},{"id":"deepseek-prover-v2-2025","title":"DeepSeek-Prover-V2: Advancing Formal Mathematical Reasoning via Reinforcement Learning for Subgoal Decomposition","year":2025,"venue":"arXiv preprint","authors":["Z. Z. Ren","Zhihong Shao","Junxiao Song","Huajian Xin","Haocheng Wang","Wanjia Zhao","Liyue Zhang","Zhe Fu","Qihao Zhu","Dejian Yang","Z. F. Wu","Zhibin Gou","Shirong Ma","Hongxuan Tang","Yuxuan Liu","Wenjun Gao","Daya Guo","Chong Ruan"],"authors_zh":"Z. Z. Ren 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","formal_theorem_proving","lean4","subgoal_decomposition"],"tags":["deepseek-prover-v2","lean4","formal-reasoning","subgoal-decomposition","grpo","rlvr","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿形式化推理技术报告与数据披露账本","best_for_zh":"需要审计 Lean 终态验证、子目标分解 RL、形式化语义边界与训练可复现性的读者","confidence":"medium","one_line":["DeepSeek-Prover-V2 combines DeepSeek-V3 proof-sketch decomposition, recursively solved Lean subgoals, Lean-verified cold starts, expert iteration, and GRPO RL, while releasing models and evaluation artifacts but not the core training or audit corpus.","DeepSeek-Prover-V2 将 DeepSeek-V3 证明草图分解、递归求解的 Lean 子目标、Lean 验证冷启动、expert iteration 和 GRPO RL 结合起来；它公开模型和评测工件，但未公开核心训练或审计语料。"],"why":"It makes a frontier formal-RL pipeline legible while separating programmatic proof success from missing source lineage, environment reproducibility, semantic-formalization audit, and data-rights evidence.","primary_link":"https://arxiv.org/abs/2504.21801","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/deepseek-ai/DeepSeek-ProverBench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/deepseek-ai/DeepSeek-Prover-V2-671B"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/deepseek-ai/DeepSeek-Prover-V2"}],"link_count":6,"sections":9},{"id":"deepseek-r1-2025","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["DeepSeek-AI"],"authors_zh":"DeepSeek-AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","coding","science","reasoning","general","safety"],"tags":["deepseek-r1","frontier-report","data-disclosure-ledger","grpo","rlvr","cold-start","rejection-sampling","distillation","reward-modeling"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型报告与数据披露账本","best_for_zh":"希望审计 GRPO、cold start、rejection sampling、蒸馏与开放权重复现边界的研究者","confidence":"high","one_line":["DeepSeek-R1 reports cold-start/RL/SFT/RL stages, rule/reward feedback, and 800K-row distillation while withholding prompts, traces, verifiers, rewards, splits, and audit artifacts.","DeepSeek-R1 报告了 cold-start/RL/SFT/RL 阶段、规则/奖励反馈和 80 万条蒸馏数据，却未公开 prompt、轨迹、verifier、reward、切分或审计 artifact。"],"why":"It separates public weights and stage narrative from unavailable data lineage, reward calibration, terminal predicates, and reproducibility evidence.","primary_link":"https://arxiv.org/abs/2501.12948","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deepseek-ai/DeepSeek-R1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/deepseek-ai/DeepSeek-R1"},{"key":"project","label":["Project","项目主页"],"url":"https://api-docs.deepseek.com/news/news250120/"}],"link_count":6,"sections":9},{"id":"deepseek-v3-1-release-2025","title":"DeepSeek-V3.1","year":2025,"venue":"Official DeepSeek Hugging Face model release","authors":["DeepSeek-AI"],"authors_zh":"DeepSeek-AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","coding"],"tags":["deepseek-v3-1","frontier-report","data-disclosure-ledger","model-release","long-context"],"status":"partial","priority":"可读","paper_type_zh":"前沿模型发布","best_for_zh":"审计版本化发布的披露边界","confidence":"medium","one_line":["DeepSeek-V3.1 releases weights, interface formats, 128K context, and base-extension token counts, but not its post-training data, feedback, or audit stack.","DeepSeek-V3.1 发布了权重、接口、128K 上下文和基础延长 token 数，但未公开后训练数据、反馈或审计栈。"],"why":"It separates a usable model release from the missing evidence required to audit how reasoning and tool use were post-trained.","primary_link":"https://huggingface.co/deepseek-ai/DeepSeek-V3.1","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deepseek-ai/DeepSeek-V3"}],"link_count":2,"sections":9},{"id":"deepseek-v3-2-pushing-frontier-2025","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","year":2025,"venue":"arXiv preprint","authors":["DeepSeek-AI"],"authors_zh":"DeepSeek-AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","agent_environment"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["full_episode","answer_level","scalar_reward"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["reasoning","coding","mathematics","agentic_tool_use"],"tags":["deepseek-v3-2","frontier-report","data-disclosure-ledger","agentic-task-synthesis","grpo","thinking-tool-use"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理与智能体模型报告和数据披露账本","best_for_zh":"审计合成 agentic 训练、GRPO、思考工具调用与开放权重背后的未公开数据和反馈边界","confidence":"high","one_line":["DeepSeek-V3.2 reports 85,267 agent prompts, 1,827 synthetic tool environments, specialist distillation, and mixed GRPO, while releasing model weights and inference code but not training records, environments, or reward artifacts.","DeepSeek-V3.2 报告 scalable GRPO、专家蒸馏以及覆盖 1,800+ 环境、85K+ 指令的合成 agentic 训练；数据、环境、奖励契约和审计资产未公开。"],"why":"It lets the frontier-report track compare unusually specific task and feedback descriptions against the much narrower set of artifacts that can actually be downloaded and replayed.","primary_link":"https://arxiv.org/abs/2512.02556","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deepseek-ai/DeepSeek-V3.2-Exp"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2"},{"key":"project","label":["Project","项目主页"],"url":"https://api-docs.deepseek.com/news/news251201/"}],"link_count":6,"sections":9},{"id":"deepseekmath-v2-2025","title":"DeepSeekMath-V2: Towards Self-Verifiable Mathematical Reasoning","year":2025,"venue":"arXiv preprint","authors":["Zhihong Shao","Yuxiang Luo","Chengda Lu","Z.Z. Ren","Jiewen Hu","Tian Ye","Zhibin Gou","Shirong Ma","Xiaokang Zhang"],"authors_zh":"Zhihong Shao、Yuxiang Luo、Chengda Lu、Z.Z. Ren、Jiewen Hu、Tian Ye、Zhibin Gou、Shirong Ma、Xiaokang Zhang","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward","process_reward"],"training_use":["sft","reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["mathematics","natural_language_theorem_proving","olympiad_reasoning","proof_verification"],"tags":["deepseek","deepseekmath-v2","natural-language-proofs","proof-verifier","meta-verification","self-verification","grpo","reward-model","automated-labeling","generator-verifier-loop","test-time-compute","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"自然语言数学证明生成—验证前沿报告与数据披露账本","best_for_zh":"研究自然语言 proof verifier、meta-verification、GRPO 数据飞轮和测试时搜索审计的读者","confidence":"high","one_line":["DeepSeekMath-V2 alternates GRPO-trained proof verification and generation, bootstraps harder proof labels through multi-sample verifier and meta-verifier judgments, and scales self-verification at test time, while releasing model weights and evaluation artifacts but not the training records or formal correctness certificates.","DeepSeekMath-V2 以专家三级标签训练自然语言 proof verifier 和 meta-verifier，再通过多采样自动标注与 GRPO 形成生成—验证闭环；其价值是披露反馈型数据飞轮，但 LLM 判断与 64 次一致通过都不是 Lean/Isabelle 形式证明证书。"],"why":"It is an unusually concrete frontier disclosure of how proof problems, generated proofs, expert scores, verifier analyses, meta-verification, learned rewards, and test-time search form a data flywheel, while exposing the audit gap between LLM agreement and formally verified mathematics.","primary_link":"https://arxiv.org/abs/2511.22570","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deepseek-ai/DeepSeek-Math-V2"},{"key":"data","label":["Data","数据"],"url":"https://github.com/deepseek-ai/DeepSeek-Math-V2/tree/main/outputs"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/deepseek-ai/DeepSeek-Math-V2"}],"link_count":5,"sections":9},{"id":"defects4c-c-cpp-bug-repair-2025","title":"Defects4C: Benchmarking Large Language Model Repair Capability with C/C++ Bugs","year":2025,"venue":"ASE 2025","authors":["Jian Wang","Xiaofei Xie","Qiang Hu","Shangqing Liu","Jiongchi Yu","Jiaolong Kong","Yi Li"],"authors_zh":"Jian Wang, Xiaofei Xie, Qiang Hu, Shangqing Liu, Jiongchi Yu, Jiaolong Kong, Yi Li","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","program-repair","software-security"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"C/C++ 缺陷修复与漏洞修复可执行基准","best_for_zh":"需要评测 C/C++ 自动程序修复、漏洞修复或基于测试的代码奖励的研究者。","confidence":"high","one_line":["Defects4C provides executable C/C++ repair tasks with 248 real bugs, 102 vulnerabilities, and reproduction tests for validating patches.","Defects4C 从真实 C/C++ 仓库整理 248 个缺陷与 102 个漏洞，配套可复现测试以执行验证修复补丁。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://arxiv.org/abs/2510.11059","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/defects4c/defects4c"},{"key":"data","label":["Data","数据"],"url":"https://github.com/defects4c/defects4c/tree/master/defectsc_tpl"},{"key":"project","label":["Project","项目主页"],"url":"https://sites.google.com/view/defects4c/"}],"link_count":4,"sections":9},{"id":"deliberate-reasoning-structure-aware-planning-accurate-world-model-2025","title":"Deliberate Reasoning in Language Models as Structure-Aware Planning with an Accurate World Model","year":2025,"venue":"ACL 2025","authors":["Siheng Xiong","Ali Payani","Yuan Yang","Faramarz Fekri"],"authors_zh":"Siheng Xiong、Ali Payani、Yuan Yang、Faramarz Fekri","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["formal_reasoning","planning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"结构感知规划、蕴含图验证与过程监督数据论文","best_for_zh":"研究带状态图的多步推理、过程判别与可验证轨迹规划。","confidence":"high","one_line":["SWAP releases trajectory and process-supervision data for structure-aware reasoning that updates and verifies entailment graphs during planning.","SWAP 发布结构感知推理的轨迹与过程监督数据，在规划中更新并验证蕴含图。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://aclanthology.org/2025.acl-long.1540/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xiongsiheng/SWAP"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sxiong/SWAP"}],"link_count":5,"sections":9},{"id":"multilingual-process-reward-modeling-emnlp-2025","title":"Demystifying Multilingual Reasoning in Process Reward Modeling","year":2025,"venue":"Findings of EMNLP 2025","authors":["Weixuan Wang","Minghao Wu","Barry Haddow","Alexandra Birch"],"authors_zh":"Wang et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Demystifying Multilingual Reasoning in Process Reward Modeling exposes process or trace supervision data.","将 PRM800K 与 Math-Shepherd 翻译为七种训练语言，系统检验多语过程奖励模型在 11 种语言数学推理上的迁移。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://aclanthology.org/2025.findings-emnlp.519/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/weixuan-wang123/Multilingual-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/vicky23456/multilingual-PRM800K"}],"link_count":3,"sections":9},{"id":"detecting-benchmark-contamination-through-watermarking-2025","title":"Detecting Benchmark Contamination Through Watermarking","year":2025,"venue":"ICLR 2025 Workshop on GenAI Watermarking","authors":["Tom Sander","Pierre Fernandez","Saeed Mahloujifar","Alain Durmus","Chuan Guo"],"authors_zh":"Tom Sander, Pierre Fernandez, Saeed Mahloujifar, Alain Durmus, Chuan Guo","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","benchmark-contamination"],"status":"verified","priority":"可读","paper_type_zh":"基准污染检测与评测可靠性审计论文","best_for_zh":"在发布前设计可审计基准、且可访问待测模型权重的维护者。","confidence":"high","one_line":["Proactive benchmark watermarking produces a calibrated white-box test for memorized contamination.","发布前给基准题目加秘密水印，并用有确定假阳性率的白盒统计检验发现模型是否记住该基准。"],"why":"It makes benchmark contamination auditable before release without relying on a clean reference set.","primary_link":"https://arxiv.org/abs/2502.17259","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/meta-seal"}],"link_count":3,"sections":9},{"id":"rl-posttraining-contamination-2025","title":"Detecting Data Contamination from Reinforcement Learning Post-training for Large Language Models","year":2025,"venue":"ICLR 2026","authors":[],"authors_zh":"Yongding Tao, Tian Wang, Yihong Dong, Huanyu Liu, Kechi Zhang, Xiaolong Hu, Ge Li","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"污染、验证器失效或奖励投机审计论文","best_for_zh":"需要审计基准污染、评估偏差或奖励投机风险的研究者。","confidence":"high","one_line":["RL-MIA benchmark and post-training contamination detector","Self-Critique以初始与批评回复的熵轨迹相似度检测 RL 后训练数据污染。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://openreview.net/forum?id=EjiJmiA6ea","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yongding-tao/RL-Data-Contamination"}],"link_count":3,"sections":9},{"id":"djpo-foundational-judge-data-2025","title":"Direct Judgement Preference Optimization","year":2025,"venue":"EMNLP 2025 Main Conference","authors":["Peifeng Wang","Austin Xu","Yilun Zhou","Caiming Xiong","Shafiq Joty"],"authors_zh":"Peifeng Wang, Austin Xu, Yilun Zhou, Caiming Xiong, Shafiq Joty","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","evaluation"],"tags":["judge-model","preference-data","dpo","critique","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"基础评审器偏好数据构造研究","best_for_zh":"从异质评测标注构建可复用大语言模型评审器或评审器奖励信号的读者。","confidence":"high","one_line":["DJPO builds three complementary DPO pair types to train generalized LLM judges from critiques, direct decisions, and response deductions.","DJPO 构造三类互补的 DPO 样本对，从批评、直接决策和回答反推训练可泛化的大语言模型评审器。"],"why":"It turns judgment labels and generated critiques into structured preference data that train a judge to evaluate, explain, and understand responses.","primary_link":"https://aclanthology.org/2025.emnlp-main.103/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SalesforceAIResearch/sfrjudge"}],"link_count":3,"sections":9},{"id":"distributional-llm-judge-2025","title":"Distributional LLM-as-a-Judge","year":2025,"venue":"NeurIPS 2025 (Poster)","authors":["Luyu Chen","Zeyu Zhang","Haoran Tan","Quanyu Dai","Hao Yang","Zhenhua Dong","Xu Chen"],"authors_zh":"Luyu Chen, Zeyu Zhang, Haoran Tan, Quanyu Dai, Hao Yang, Zhenhua Dong, Xu Chen","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"分布式 LLM 判别训练方法","best_for_zh":"需要保留人类评审分歧与不确定性的自动评测研究者。","confidence":"high","one_line":["Aligns distributions of automated verdicts with human evaluation distributions.","让 LLM 输出与人类分歧结构对齐的分布式判别器。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=0SRGbRbngJ","links":[],"link_count":1,"sections":9},{"id":"diverse-not-short-selection-2025","title":"Diverse, not Short: A Length-Controlled Data Selection Strategy for Improving Response Diversity of Language Models","year":2025,"venue":"EMNLP 2025 Main Conference","authors":["Vijeta Deshpande","Debasmita Ghose","John D. Patterson","Roger Beaty","Anna Rumshisky"],"authors_zh":"Vijeta Deshpande, Debasmita Ghose, John D. Patterson, Roger Beaty, Anna Rumshisky","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["creative-generation","alignment"],"tags":["preference-data","dpo","data-selection","diversity","length-control"],"status":"verified","priority":"必读","paper_type_zh":"长度控制的偏好数据选择研究","best_for_zh":"为开放式生成构造多样化偏好数据且需要防范指标驱动的长度偏差的读者。","confidence":"high","one_line":["Diverse-NS creates DPO pairs whose chosen response is more diverse and better scored than the rejected response without exploiting shorter length.","Diverse-NS 构造用于 DPO 的偏好对：入选回答更具多样性且质量更高，同时不靠缩短文本取得优势。"],"why":"It shows that a diversity objective changes the training record in a harmful way unless selection explicitly equalizes response length.","primary_link":"https://aclanthology.org/2025.emnlp-main.1721/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/text-machine-lab/diverse-not-short"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/text-machine-lab/diverse-not-short"}],"link_count":4,"sections":9},{"id":"contextualjudgebench-contextual-llm-judges-2025","title":"Does Context Matter? ContextualJudgeBench for Evaluating LLM-based Judges in Contextual Settings","year":2025,"venue":"ACL 2025","authors":["Austin Xu","Srijan Bansal","Yifei Ming","Semih Yavuz","Shafiq Joty"],"authors_zh":"Austin Xu、Srijan Bansal、Yifei Ming、Semih Yavuz、Shafiq Joty","tracks":["preference_reward_feedback_data","judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["rag","summarization","contextual-evaluation"],"tags":["llm-as-a-judge","rag","summarization","acl-2025"],"status":"verified","priority":"可读","paper_type_zh":"上下文条件 LLM-as-a-Judge 基准","best_for_zh":"需要 RAG、摘要或上下文依赖评审数据的研究者","confidence":"high","one_line":["ContextualJudgeBench supplies 2,000 context-dependent response pairs that test whether LLM judges apply conditional factuality, completeness, refusal, and conciseness criteria.","ContextualJudgeBench 提供约 2K RAG 与摘要比较样本，以忠实性、完整性等上下文条件评测 judge 的推理反馈质量。"],"why":"It makes explicit the evidence and conditional criteria a judge must use, rather than scoring isolated answers.","primary_link":"https://aclanthology.org/2025.acl-long.470/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SalesforceAIResearch/ContextualJudgeBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Salesforce/ContextualJudgeBench"}],"link_count":4,"sections":9},{"id":"detection-assumptions-2025","title":"Does Data Contamination Detection Work (Well) for LLMs? A Survey and Evaluation on Detection Assumptions","year":2025,"venue":"Findings of NAACL 2025","authors":["Yujuan Velvin Fu","Özlem Uzuner","Meliha Yetisgen","Fei Xia"],"authors_zh":"Yujuan Velvin Fu, Özlem Uzuner, Meliha Yetisgen, Fei Xia","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"污染检测方法综述与假设审计","best_for_zh":"需要选择或解释 LLM 污染检测结果的评测研究者。","confidence":"high","one_line":["Accepted evaluation of contamination-detection assumptions with public proceedings materials.","从假设层面审计数据污染检测，揭示实例级 MIA 的适用边界。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.findings-naacl.291/","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/2025.findings-naacl.291.pdf"}],"link_count":2,"sections":9},{"id":"thinking-more-mirage-tts-2025","title":"Does Thinking More Always Help? Mirage of Test-Time Scaling in Reasoning Models","year":2025,"venue":"NeurIPS 2025","authors":["Soumya Suvra Ghosal","Souradip Chakraborty","Avinash Reddy","Yifu Li","Mengdi Wang","Dinesh Manocha","Furong Huang","Mohammad Ghavamzadeh","Amrit Singh Bedi"],"authors_zh":"Soumya Suvra Ghosal、Souradip Chakraborty、Avinash Reddy、Yifu Li、Mengdi Wang、Dinesh Manocha、Furong Huang、Mohammad Ghavamzadeh、Amrit Singh Bedi（机构：马里兰大学、普林斯顿大学、Capital One、Amazon AGI、中央佛罗里达大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","reasoning","overthinking","parallel-thinking","budget-allocation"],"status":"verified","priority":"必读","paper_type_zh":"测试时计算扩展分析与预算分配研究（NeurIPS 2025）","best_for_zh":"需要在长链推理与多路径推理之间设计推理预算的读者。","confidence":"high","one_line":["The paper diagnoses overthinking under sequential test-time scaling and compares it with parallel allocation of the same reasoning-token budget.","该论文揭示单轨迹延长思考的过度思考拐点，并以并行推理比较同一 token 预算的更稳健分配。"],"why":"It turns the vague instruction to think longer into an explicit, measurable budget-allocation decision with a documented failure regime.","primary_link":"https://arxiv.org/abs/2506.04210","links":[],"link_count":3,"sections":9},{"id":"dolomites-domain-specific-long-form-tasks-2025","title":"Dolomites: Domain-Specific Long-Form Methodical Tasks","year":2025,"venue":"Transactions of the Association for Computational Linguistics","authors":["Chaitanya Malaviya","Priyanka Agrawal","Kuzman Ganchev","Pranesh Srinivasan","Fantine Huot","Jonathan Berant","Mark Yatskar","Dipanjan Das","Mirella Lapata","Chris Alberti"],"authors_zh":"Chaitanya Malaviya、Priyanka Agrawal、Kuzman Ganchev、Pranesh Srinivasan、Fantine Huot、Jonathan Berant、Mark Yatskar、Dipanjan Das、Mirella Lapata、Chris Alberti","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["safety","factuality-grounding"],"tags":["track7","judgment-feedback"],"status":"verified","priority":"可读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["DoLoMiTes is the paper's released feedback or evaluation resource.","收集 25 个领域专家提出的 519 类长文方法任务及 1,857 份专家修订，保留真实输入、输出与修订判断，适合通用长程推理生成。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2025.tacl-1.1/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/cmalaviya/dolomites"},{"key":"project","label":["Project","项目主页"],"url":"https://chaitanyamalaviya.github.io/data/"}],"link_count":4,"sections":9},{"id":"fetch-2025","title":"Don't Get Lost in the Trees: Streamlining LLM Reasoning by Overcoming Tree Search Exploration Pitfalls","year":2025,"venue":"ACL 2025","authors":["Ante Wang","Linfeng Song","Ye Tian","Dian Yu","Haitao Mi","Xiangyu Duan","Zhaopeng Tu","Jinsong Su","Dong Yu"],"authors_zh":"Ante Wang、Linfeng Song、Ye Tian、Dian Yu、Haitao Mi、Xiangyu Duan、Zhaopeng Tu、Jinsong Su、Dong Yu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","scalar_reward","trajectory_value"],"training_use":["process_supervision","test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["mathematics","reasoning"],"tags":["fetch","tree-search","state-merging","verifier-ensemble","td-learning","test-time-compute"],"status":"partial","priority":"必读","paper_type_zh":"推理时树搜索与过程价值建模研究","best_for_zh":"需要审计测试时搜索轨迹、状态合并和 verifier 反馈的读者","confidence":"high","one_line":["FETCH reduces redundant tree expansion by clustering equivalent reasoning states and stabilizes verifier-guided search with TD-trained, ensembled scores.","FETCH 合并语义等价推理状态，并以 TD(λ) 训练且集成的 verifier 分数引导树搜索，从而减少冗余扩展与评分方差。"],"why":"It identifies the internal trace objects and score variance that determine whether additional test-time search compute is useful, while exposing missing rollout and label-release evidence.","primary_link":"https://aclanthology.org/2025.acl-long.1167/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XMUDeepLIT/Fetch"}],"link_count":4,"sections":9},{"id":"dorm-preference-data-weights-2025","title":"DORM: Preference Data Weights Optimization for Reward Modeling in LLM Alignment","year":2025,"venue":"Findings of EMNLP 2025","authors":["Rongzhi Zhang","Chenwei Zhang","Xinyang Zhang","Liang Qiu","Haoming Jiang","Yuchen Zhuang","Qingru Zhang","Hyokun Yun","Xian Li","Bing Yin","Tuo Zhao","Chao Zhang"],"authors_zh":"Rongzhi Zhang, Chenwei Zhang, Xinyang Zhang, Liang Qiu, Haoming Jiang, Yuchen Zhuang, Qingru Zhang, Hyokun Yun, Xian Li, Bing Yin, Tuo Zhao, Chao Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","reward_modeling","preference_data"],"tags":["reward-modeling","preference-data","data-reweighting","bilevel-optimization","alignment"],"status":"verified","priority":"必读","paper_type_zh":"用于奖励模型的偏好数据重加权研究","best_for_zh":"适合混合多种偏好来源，并希望在奖励模型或偏好学习训练前以原则化方法替代硬筛选的读者。","confidence":"high","one_line":["DORM learns validation-regularized weights for heterogeneous preference data so reward-model training emphasizes informative uncertainty and downweights likely noise.","DORM 为异构偏好数据学习由验证集正则化的权重，使奖励模型训练强调有信息量的不确定样本并降低可能噪声的影响。"],"why":"It makes data importance a learned training object and carries that decision through reward modeling to downstream policy alignment.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1237/","links":[],"link_count":2,"sections":9},{"id":"dots-dynamic-reasoning-2025","title":"DOTS: Learning to Reason Dynamically in LLMs via Optimal Reasoning Trajectories Search","year":2025,"venue":"ICLR 2025","authors":["Murong Yue","Wenlin Yao","Haitao Mi","Dian Yu","Ziyu Yao","Dong Yu"],"authors_zh":"Murong Yue、Wenlin Yao、Haitao Mi、Dian Yu、Ziyu Yao、Dong Yu","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","trajectory_value"],"training_use":["sft","test_time_compute"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer"],"domains":["mathematics","symbolic_reasoning","commonsense_reasoning","scientific_qa","reading_comprehension"],"tags":["dots","optimal-reasoning-trajectories","reasoning-action-search","dynamic-reasoning","solver-conditioned-planner","raw-rollouts","rejected-traces","outcome-verification","supervised-fine-tuning","test-time-compute"],"status":"partial","priority":"必读","paper_type_zh":"推理动作轨迹搜索配方与原始 scored rollout 数据发布","best_for_zh":"研究搜索生成推理数据、轨迹级 outcome feedback、动态 test-time compute、失败样本复用及 paper-code 配置审计的读者","confidence":"high","one_line":["DOTS searches 12 solver-conditioned reasoning-action trajectories with repeated ground-truth scoring, trains external or internalized planners on the selected paths, and releases a 4.8 GB MATH file containing raw dialogues, failed trials, trajectory IDs, predicted answers, and binary scores.","DOTS 对 12 条三层推理动作轨迹进行 solver 条件化重复搜索和真值评分，用选中路径构造 planner SFT 数据；公开的 4.8 GB MATH 原始 JSON 保留成功与失败对话、轨迹、预测答案和二值分数，但缺少稳定计数、splits、剪枝清单、最终 SFT 数据与 checkpoints。"],"why":"The work makes trajectory choice and search budget first-class reasoning-data variables and exposes more raw search evidence than a final-only SFT release, while showing why solver/evaluator overfit, missing pruning manifests, and paper-code configuration drift must be audited before reuse.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/5e5d6f9ac33ba9349ba7b2be9f21bad9-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MurongYue/DOTS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MurongYue/DOTS_searching_results"}],"link_count":5,"sections":9},{"id":"dr-tulu-2025","title":"DR Tulu: Reinforcement Learning with Evolving Rubrics for Deep Research","year":2025,"venue":"ICML 2026","authors":["Rulin Shao","Akari Asai","Shannon Zejiang Shen","Hamish Ivison","Varsha Kishore","Jingming Zhuo","Xinran Zhao","Molly Park","Samuel G. Finlayson","David Sontag","Tyler Murray","Sewon Min","Pradeep Dasigi","Luca Soldaini","Faeze Brahman","Wen-tau Yih","Tongshuang Wu","Luke Zettlemoyer","Yoon Kim","Hannaneh Hajishirzi","Pang Wei Koh"],"authors_zh":"Rulin Shao、Akari Asai、Shannon Zejiang Shen、Hamish Ivison、Varsha Kishore、Jingming Zhuo、Xinran Zhao、Molly Park、Samuel G. Finlayson、David Sontag、Tyler Murray、Sewon Min、Pradeep Dasigi、Luca Soldaini、Faeze Brahman、Wen-tau Yih、Tongshuang Wu、Luke Zettlemoyer、Yoon Kim、Hannaneh Hajishirzi、Pang Wei Koh","tracks":["frontier_reports_data_disclosure_ledger","judgment_rubric_domain_expert_data"],"source_role":["construction_recipe","data_release","verifier_reward","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["deep_research","information_seeking","science","healthcare","general"],"tags":["deep-research","agent-training","evolving-rubrics","llm-judge","citations","tool-use","grpo","asynchronous-rl","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"深度研究智能体的演化 rubric 强化学习配方、数据发布边界与反馈审计","best_for_zh":"关注智能体后训练、LLM 评判、检索与引用反馈、数据血缘和可复现实验环境的研究者与工程人员","confidence":"high","one_line":["Releases SFT/RL datasets, code and checkpoints for a deep-research agent trained with GRPO and search-grounded evolving rubrics, while the final data mixture, online rubric history and evaluator calibration remain only partially disclosed.","DR-Tulu 发布了深度研究智能体的 SFT/RL 数据、代码和检查点；其 GRPO 训练结合检索锚定、持续与演化的自然语言 rubric 以及程序化协议奖励，但最终数据混合、在线 rubric 历史与评判校准仍未完整披露。"],"why":"Makes a long-form, tool-using feedback contract concrete enough to audit: it separates programmatic protocol rewards from LLM-judged rubrics and shows why an open model/data release still needs per-step rubric, source-rights and environment provenance before reuse.","primary_link":"https://arxiv.org/abs/2511.19399","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rlresearch/dr-tulu"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/rl-research/dr-tulu-rl-data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/rl-research/dr-tulu"},{"key":"project","label":["Project","项目主页"],"url":"https://allenai.org/blog/dr-tulu"}],"link_count":6,"sections":9},{"id":"dynamic-early-exit-reasoning-2025","title":"Dynamic Early Exit in Reasoning Models","year":2025,"venue":"ICLR 2026","authors":["Chenxu Yang","Qingyi Si","Yongjie Duan","Zheliang Zhu","Chenyu Zhu","Qiaowei Li","Minghui Chen","Zheng Lin","Weiping Wang"],"authors_zh":"Chenxu Yang、Qingyi Si、Yongjie Duan、Zheliang Zhu、Chenyu Zhu、Qiaowei Li、Minghui Chen、Zheng Lin、Weiping Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["scaling_study","construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute"],"construction_layer":["scaling_report"],"domains":["mathematics","science","code"],"tags":["deer","early-exit","self-confidence","test-time-compute"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["DEER probes trial answers at reasoning transitions and exits when model self-confidence is high, while the official repository releases only part of the paper's implementation and no reusable rollout corpus.","DEER 在推理的转折处试探性给出答案，当模型自信度足够高时提前退出；官方仓库只发布了论文实现的一部分，且没有可复用的采样语料。"],"why":"It makes adaptive trace truncation auditable as a sequence of transition detection, answer induction, confidence estimation, and branch selection rather than treating reduced benchmark token counts as data-quality evidence.","primary_link":"https://arxiv.org/abs/2504.15895","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/iie-ycx/DEER"}],"link_count":5,"sections":9},{"id":"dynamic-scaling-unit-tests-code-reward-modeling-2025","title":"Dynamic Scaling of Unit Tests for Code Reward Modeling","year":2025,"venue":"ACL 2025 (Long Papers)","authors":["Zeyao Ma","Xiaokang Zhang","Jing Zhang","Jifan Yu","Sijia Luo","Jie Tang"],"authors_zh":"Zeyao Ma, Xiaokang Zhang, Jing Zhang, Jifan Yu, Sijia Luo, Jie Tang","tracks":["rollout_search_test_time_trace_data","audit_failure_contamination_verifier_attacks"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","programmatic_code_formal"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["Dynamic Scaling of Unit Tests for Code Reward Modeling records candidate solutions generated unit tests execution rewards and test budgets under test-derived reward.","CodeRM 发布合成单元测试与错误率字段，并按题目难度动态分配测试时验证预算。"],"why":"It makes difficulty-adaptive number of tests and BoN selection and its audit boundary visible for reasoning-data curation.","primary_link":"https://aclanthology.org/2025.acl-long.343/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCKBReasoning/CodeRM"}],"link_count":3,"sections":9},{"id":"dyve-2025","title":"Dyve: Thinking Fast and Slow for Dynamic Process Verification","year":2025,"venue":"EMNLP 2025","authors":["Jianyuan Zhong","Zeju Li","Zhijian Xu","Xiangyu Wen","Qiang Xu"],"authors_zh":"Jianyuan Zhong, Zeju Li, Zhijian Xu, Xiangyu Wen, Qiang Xu","tracks":["data_construction_open_release_recipes","preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["verifier_reward","process_supervision","construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","full_episode"],"training_use":["sft","reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics"],"tags":["process-verifier","process-supervision","consensus-filtering","omega-prm","mcts-rollouts","llm-as-judge","system-1-system-2","first-error-detection","synthetic-reasoning-data","open-release"],"status":"partial","priority":"必读","paper_type_zh":"生成式 process verifier、监督配方与开放发布","best_for_zh":"构造 process-reward data、训练 step verifier、过滤搜索轨迹，或研究验证精度与推理成本权衡的研究者","confidence":"high","one_line":["An open process-supervision recipe that filters noisy MCTS traces and converts retained steps into mixed one-token and explanatory verifier targets.","这是一个开放的 process-supervision 配方：它过滤带噪 MCTS 轨迹，并把保留步骤转换成单 token 快速判决或带分析的慢速验证 target。"],"why":"The work exposes a reusable verifier-data refresh pipeline while making judge correlation, rebalancing, release mismatch, and license risks central to reuse decisions.","primary_link":"https://aclanthology.org/2025.emnlp-main.1136/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/staymylove/Dyve"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Jianyuan1/cot-data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Jianyuan1/deepseek-r1-14b-cot-math-reasoning-full"}],"link_count":6,"sections":9},{"id":"e3-test-time-compute-exploration-2025","title":"e3: Learning to Explore Enables Extrapolation of Test-Time Compute for LLMs","year":2025,"venue":"arXiv","authors":["Amrith Setlur","Matthew Y. R. Yang","Charlie Snell","Jeremy Greer","Ian Wu","Virginia Smith","Max Simchowitz","Aviral Kumar"],"authors_zh":"Amrith Setlur、Matthew Y. R. Yang、Charlie Snell、Jeremy Greer、Ian Wu、Virginia Smith、Max Simchowitz、Aviral Kumar","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematics","test_time_compute"],"tags":["e3","test-time-compute","in-context-exploration","grpo","outcome-reward-rl","negative-gradients","curriculum","math-reasoning","rollout-budget"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹、搜索或测试时计算研究","best_for_zh":"需要审计 rollout、选择器、预算、数据谱系与复现边界的读者","confidence":"medium","one_line":["e3 releases staged math prompt/ground-truth data, code, and a Qwen3-1.7B checkpoint for an 8k-to-16k GRPO curriculum, but not the grouped on-policy rollouts or per-trace advantages used during training.","该条目将推理轨迹、搜索选择或测试时计算作为可审计的研究对象；未披露字段已明确标为 unknown。"],"why":"For rollout and test-time trace data, e3 makes group composition, failed-response gradients, task difficulty, and token budget part of the training-data contract; without the grouped traces, its central feedback mechanism cannot be fully replayed or audited.","primary_link":"https://arxiv.org/abs/2506.09026","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ars22/e3"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/CMU-AIRe/e3-math-easy"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/CMU-AIRe/e3-1.7B"},{"key":"project","label":["Project","项目主页"],"url":"https://matthewyryang.com/e3/"}],"link_count":8,"sections":9},{"id":"llama-nemotron-vlm-data-2025","title":"Eagle 2: Building Post-Training Data Strategies from Scratch for Frontier Vision-Language Models","year":2025,"venue":"arXiv preprint","authors":["Zhiqi Li","Guo Chen","Shilong Liu","Shihao Wang","Vibashan VS","Yishen Ji","Shiyi Lan","Hao Zhang","Yilin Zhao","Subhashree Radhakrishnan","Nadine Chang","Karan Sapra","Amala Deshmukh","Tuomas Rintamaki","Matthieu Le","De-An Huang","Ilia Karmanov","Lukas Voegtle","Philipp Fischer","Timo Roman","Tong Lu","Jose M. Alvarez","Bryan Catanzaro","Jan Kautz","Andrew Tao","Guilin Liu","Zhiding Yu"],"authors_zh":"Zhiqi Li、Guo Chen、Shilong Liu、Shihao Wang、Vibashan VS、Yishen Ji、Shiyi Lan、Hao Zhang、Yilin Zhao、Subhashree Radhakrishnan、Nadine Chang、Karan Sapra、Amala Deshmukh、Tuomas Rintamaki、Matthieu Le、De-An Huang、Ilia Karmanov、Lukas Voegtle、Philipp Fischer、Timo Roman、Tong Lu、Jose M. Alvarez、Bryan Catanzaro、Jan Kautz、Andrew Tao、Guilin Liu、Zhiding Yu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["data-centric-frontier-VLM-post-training"],"tags":["instruction-demonstration-rationale","arxiv-2501.14818","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"前沿视觉语言模型的后训练监督微调","confidence":"high","one_line":["Eagle 2 studies each data strategy from scratch and releases the resulting capability-labeled conversation mixture with per-source licensing metadata.","Eagle 2 从来源选择、改写、配比和过滤开始重建视觉语言后训练数据，并公开带来源信息的多能力对话混合。"],"why":"Frontier VLM reports rarely disclose how raw multimodal sources are selected, reformatted, balanced, and filtered into a competitive post-training mixture.","primary_link":"https://arxiv.org/abs/2501.14818","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Llama-Nemotron-VLM-Dataset-v1"}],"link_count":2,"sections":9},{"id":"effibench-x-multilanguage-code-efficiency-2025","title":"EffiBench-X: A Multi-Language Benchmark for Measuring Efficiency of LLM-Generated Code","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Yuhao Qing","Boyu Zhu","Mingzhe Du","Zhijiang Guo","Terry Yue Zhuo","Qianru Zhang","Jie M. Zhang","Heming Cui","Siu-Ming Yiu","Dong Huang","See-Kiong Ng","Luu Anh Tuan"],"authors_zh":"Yuhao Qing, Boyu Zhu, Mingzhe Du, Zhijiang Guo, Terry Yue Zhuo, Qianru Zhang, Jie M. Zhang, Heming Cui, Siu-Ming Yiu, Dong Huang, See-Kiong Ng, Luu Anh Tuan","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code"],"tags":["code-efficiency","multilingual","sandbox","benchmark","neurips-2025"],"status":"verified","priority":"可读","paper_type_zh":"跨语言、执行式代码效率基准","best_for_zh":"研究代码优化、效率奖励和跨语言代码生成的研究者。","confidence":"high","one_line":["EffiBench-X tests whether correct code approaches expert runtime and memory efficiency across six languages.","EffiBench-X 检验通过功能测试的代码能否在六种语言中接近专家的运行时间与内存效率。"],"why":"It turns code efficiency into a verifiable outcome rather than treating functional correctness as sufficient.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/6b8198f777d1ba87a8250368db2ea16a-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/EffiBench/EffiBench-X"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/EffiBench/effibench-x"}],"link_count":4,"sections":9},{"id":"pc-agent-e-2025","title":"Efficient Agent Training for Computer Use","year":2025,"venue":"ICLR 2026","authors":["Yanheng He","Jiahe Jin","Pengfei Liu"],"authors_zh":"Yanheng He, Jiahe Jin, Pengfei Liu","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","construction_recipe","agent_environment","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","release_audit"],"domains":["computer-use-agents","windows-gui","visual-agent-training"],"tags":["environment-agent-trajectory-data","computer-use-agent","trajectory-augmentation","multimodal-sft","synthetic-actions","WindowsAgentArena-V2","ICLR-2026"],"status":"verified","priority":"可读","paper_type_zh":"计算机使用智能体数据发布与构造配方","best_for_zh":"研究 GUI agent 轨迹构造、多模态 SFT、环境反馈契约与合成动作审计的读者","confidence":"high","one_line":["PC Agent-E turns 312 human Windows demonstrations into action-level screenshot-history-to-thought/action SFT data, but its Claude-generated branch actions are never environment-validated.","PC Agent-E 将 312 条人工 Windows 完整轨迹扩展为 action-level 多模态 SFT 数据，但每步九个 Claude 分支均未执行，也没有经过环境成功验证。"],"why":"It is a compact, released example of trajectory-to-SFT construction whose gains can be tied to a concrete data object, while clearly showing why synthetic agent actions should not be mistaken for verified environment episodes.","primary_link":"https://arxiv.org/abs/2505.13909","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/PC-Agent-E"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/henryhe0123/PC-Agent-E"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/henryhe0123/PC-Agent-E"},{"key":"project","label":["Project","项目主页"],"url":"https://gair-nlp.github.io/PC-Agent-E/"}],"link_count":9,"sections":9},{"id":"efficient-inference-lrm-survey-2025","title":"Efficient Inference for Large Reasoning Models: A Survey","year":2025,"venue":"arXiv preprint","authors":["Yue Liu","Jiaying Wu","Yufei He","Hongcheng Gao","Hongyu Chen","Baolong Bi","Jiaheng Zhang","Zhiqi Huang","Bryan Hooi"],"authors_zh":"Yue Liu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["test_time_compute","evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","chain-of-thought","inference-efficiency","reasoning-data","test-time-compute"],"tags":["foundations-and-primers","efficient-inference","large-reasoning-models","nus","survey"],"status":"verified","priority":"必读","paper_type_zh":"大推理模型高效推理综述","best_for_zh":"在内存、延迟或 token 预算下优化长推理过程的读者。","confidence":"high","one_line":["An NUS-led survey of efficient inference methods for long reasoning models.","把高效推理分为显式紧凑过程和隐式潜在过程的 NUS 综述。"],"why":"It separates reducing visible tokens from preserving the information needed for correct reasoning.","primary_link":"https://arxiv.org/abs/2503.23077","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yueliu1999/Awesome-Efficient-Inference-for-LRMs"}],"link_count":3,"sections":9},{"id":"efficient-process-reward-model-training-active-learning-2025","title":"Efficient Process Reward Model Training via Active Learning","year":2025,"venue":"COLM 2025","authors":["Keyu Duan","Zichen Liu","Xin Mao","Tianyu Pang","Changyu Chen","Qiguang Chen","Michael Qizhe Shieh","Longxu Dou"],"authors_zh":"Keyu Duan、Zichen Liu、Xin Mao、Tianyu Pang、Changyu Chen 等","tracks":["process_trace_supervision_data","training_usage_optimization_objectives"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["math_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"主动学习式过程奖励数据构建论文","best_for_zh":"以较低标注预算训练或研究数学过程奖励模型。","confidence":"high","one_line":["ActPRM actively selects uncertain mathematical reasoning trajectories for expensive step-level labeling, releasing a 663K-record PRM dataset.","ActPRM 主动挑选不确定的数学推理轨迹进行昂贵的逐步标注，并发布约 66.3 万条 PRM 数据。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2504.10559","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sail-sg/ActivePRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sail/ActPRMData"}],"link_count":5,"sections":9},{"id":"self-calibration-efficient-tts-2025","title":"Efficient Test-Time Scaling via Self-Calibration","year":2025,"venue":"arXiv preprint; ICLR 2026","authors":["Chengsong Huang","Langlin Huang","Jixuan Leng","Jiacheng Liu","Jiaxin Huang"],"authors_zh":"Chengsong Huang、Langlin Huang、Jiaxin Huang（华盛顿大学圣路易斯分校）；Jixuan Leng（卡内基梅隆大学）；Jiacheng Liu（华盛顿大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-scaling","confidence-calibration","early-stopping"],"status":"verified","priority":"可读","paper_type_zh":"置信度校准测试时扩展研究","best_for_zh":"希望减少固定最佳候选采样成本的读者。","confidence":"high","one_line":["Self-Calibration distills self-consistency confidence into one forward pass, enabling confidence-based early stopping for sampling.","自校准将自一致性置信度蒸馏到单次前向传播，使采样可以依据置信度早停。"],"why":"It turns calibrated confidence into an online compute-allocation signal.","primary_link":"https://arxiv.org/abs/2503.00031","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Chengsong-Huang/Self-Calibration"}],"link_count":3,"sections":9},{"id":"certaindex-efficient-reasoning-2025","title":"Efficiently Scaling LLM Reasoning Programs with Certaindex","year":2025,"venue":"NeurIPS 2025","authors":["Yichao Fu","Junda Chen","Siqi Zhu","Zheyu Fu","Zhongdongming Dai","Yonghao Zhuang","Yian Ma","Aurick Qiao","Tajana Rosing","Ion Stoica","Hao Zhang"],"authors_zh":"Yichao Fu、Junda Chen、Siqi Zhu、Zheyu Fu、Zhongdongming Dai、Yonghao Zhuang、Yian Ma、Aurick Qiao、Tajana Rosing、Ion Stoica、Hao Zhang（机构：加州大学圣迭戈分校、卡内基梅隆大学、Snowflake、加州大学伯克利分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-compute","early-exit","scheduling","reasoning-efficiency"],"status":"verified","priority":"必读","paper_type_zh":"测试时推理效率与服务系统研究","best_for_zh":"部署自适应早停和推理任务调度的读者。","confidence":"high","one_line":["Certaindex detects stabilized reasoning answers to stop or reallocate inference compute without lowering accuracy.","Certaindex 检测推理答案是否已稳定，以停止或重分配计算而不降低准确率。"],"why":"It connects a model-side stability signal to practical token allocation and serving throughput.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/d037fd021c9aace128b8ce25001cdb6c-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hao-ai-lab/Dynasor"}],"link_count":3,"sections":9},{"id":"symbolically-guided-monte-carlo-process-supervision-2025","title":"Enhancing Logical Reasoning in Language Models via Symbolically-Guided Monte Carlo Process Supervision","year":2025,"venue":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (EMNLP 2025)","authors":["Xingwei Tan","Marco Valentino","Mahmud Elahi Akhter","Maria Liakata","Nikolaos Aletras"],"authors_zh":"Xingwei Tan、Marco Valentino、Mahmud Elahi Akhter、Maria Liakata、Nikolaos Aletras","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","process_supervision","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","trajectory_value","pairwise_preference"],"training_use":["sft","preference_learning","reward_modeling","process_supervision"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["logic","reasoning","claim_verification"],"tags":["symbolic-react","monte-carlo-process-supervision","process-reward-model","symbolic-reasoning","logical-reasoning","rollout-traces","pseudo-labels","sft","dpo","preference-pairs"],"status":"partial","priority":"可读","paper_type_zh":"符号推理 rollout、过程监督数据与筛选配方","best_for_zh":"研究 Monte Carlo 过程标签、PRM 筛选、SFT/DPO 数据构造及发布审计的读者","confidence":"high","one_line":["Releases 1,505 Symbolic ReAct trajectories with 12,448 Monte Carlo-derived step labels, plus 15,412 accepted SFT traces and 21,472 PRM-ranked DPO pairs derived from FOLIO, FOLIOv2, and LogicAsker.","该工作用答案可达性的 Monte Carlo 伪标签训练 PRM，将 Symbolic ReAct rollout 筛成 15,412 条 SFT 轨迹和 21,472 个 DPO 对；主要价值是公开了数据转换链，关键边界是步骤形式有效性与完整候选池仍不可审计。"],"why":"It makes the offline data transformation explicit—answer-checked rollouts become step supervision, a PRM filters and scores new traces, and those scores yield SFT and preference data—while exposing the gap between answer reachability and formally valid reasoning.","primary_link":"https://aclanthology.org/2025.emnlp-main.1624/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Xingwei-Tan/Symbolic-Guided_MC"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/XingweiT/symbreact-trace"}],"link_count":5,"sections":9},{"id":"enhancing-process-supervision-mcts-2025","title":"Enhancing Reasoning through Process Supervision with Monte Carlo Tree Search","year":2025,"venue":"AAAI 2025 NeurMAD workshop (per arXiv comments)","authors":["Shuangtao Li","Shuaihao Dong","Kexin Luan","Xinhan Di","Chaofan Ding"],"authors_zh":"Shuangtao Li；Shuaihao Dong；Kexin Luan；Xinhan Di；Chaofan Ding","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","scalar_reward"],"training_use":["sft","process_supervision"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["mcts","process-supervision","relative-scores","sft","kl-regularization"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A process-supervision recipe that uses MCTS to create scored partial-reasoning examples.","该方法用 MCTS 采样推理步骤，为步骤赋予相对正确性分数，并以加权对数似然反复训练模型生成过程监督数据。"],"why":"It makes step-level supervision lineage explicit, while showing why final binary outcomes cannot validate unshared intermediate relative scores.","primary_link":"https://arxiv.org/abs/2501.01478","links":[],"link_count":1,"sections":9},{"id":"r2-llms-retrieval-mcts-2025","title":"Enhancing Test-Time Scaling of Large Language Models with Hierarchical Retrieval-Augmented MCTS","year":2025,"venue":"arXiv preprint","authors":["Alex ZH Dou","Zhongwei Wan","Dongfei Cui","Xin Wang","Jing Xiong","Haokun Lin","Chaofan Tao","Shen Yan","Mi Zhang"],"authors_zh":"Alex ZH Dou（凯斯西储大学）；Zhongwei Wan、Xin Wang、Mi Zhang（俄亥俄州立大学）；Dongfei Cui（杜克大学）；Jing Xiong、Chaofan Tao（香港大学）；Haokun Lin（香港城市大学）；Shen Yan（字节跳动）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","optimizer_scaffold"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","mcts","retrieval","process-reward-model","mathematical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"检索增强测试时搜索研究","best_for_zh":"比较数学推理中搜索时检索与过程验证方法的读者。","confidence":"high","one_line":["R2-LLMs uses coarse-to-fine retrieval inside PRM-guided MCTS so additional test-time compute explores reasoning paths with relevant external evidence.","R2-LLMs 将问题级与步骤级检索嵌入过程奖励模型引导的蒙特卡洛树搜索，使额外测试时计算能基于相关证据探索推理路径。"],"why":"It makes retrieval an active component of the test-time compute allocation rather than a static prompt prefix.","primary_link":"https://arxiv.org/abs/2507.05557","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SUSTechBruce/R2LLM"}],"link_count":3,"sections":9},{"id":"enhancing-outcome-reward-based-rl-training-mllms-self-consistency-sampling-2025","title":"Enhancing the Outcome Reward-based RL Training of MLLMs with Self-Consistency Sampling","year":2025,"venue":"NeurIPS 2025","authors":["Jiahao Wang","Weiye Xu","Aijun Yang","Wengang Zhou","Lewei Lu","Houqiang Li","Xiaohua Wang","Jinguo Zhu"],"authors_zh":"Jiahao Wang；Weiye Xu；Aijun Yang；Wengang Zhou；Lewei Lu；Houqiang Li；Xiaohua Wang；Jinguo Zhu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["multimodal","reasoning"],"tags":["mllm","rlvr","self-consistency","resampling","outcome-reward"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A multimodal online-RL rollout/resampling recipe with official code and data links.","SCS 对视觉推理轨迹作小视觉扰动、截断和重采样，以轨迹一致性分数降低“猜对但推理错误”的结果奖励影响。"],"why":"It is valuable because it exposes where multimodal trace construction needs stronger provenance: perturbations, resampling seeds, outcome/format rewards, and dataset-count reconciliation.","primary_link":"https://arxiv.org/abs/2511.10648","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GenuineWWD/SCS"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/GenuineWWD/SCS_data"}],"link_count":3,"sections":9},{"id":"er-prm-entropy-process-reward-2025","title":"Entropy-Regularized Process Reward Model","year":2025,"venue":"Transactions on Machine Learning Research 2025","authors":["Hanning Zhang","Pengcheng Wang","Shizhe Diao","Yong Lin","Rui Pan","Hanze Dong","Dylan Zhang","Pavlo Molchanov","Tong Zhang"],"authors_zh":"Hanning Zhang、Pengcheng Wang、Shizhe Diao、Yong Lin、Rui Pan、Hanze Dong、Dylan Zhang、Pavlo Molchanov、Tong Zhang（伊利诺伊大学厄巴纳-香槟分校、多伦多大学、英伟达、普林斯顿大学、Salesforce Research）","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision","construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["process_reward"],"training_use":["process_supervision","sft","test_time_compute"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning"],"tags":["process-reward-model","entropy-regularization","rollout-labeling","rejection-sampling","mathematical-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"熵正则化过程标签构造、奖励建模与策略选择研究","best_for_zh":"适合研究如何把轨迹补全结果转化为步骤奖励，以及奖励模型如何进一步过滤训练推理链的读者。","confidence":"high","one_line":["ER-PRM derives process labels from reference-policy completion rollouts under a KL-consistent entropy formulation, then uses them to train and apply a math reward model.","ER-PRM 在与 KL 正则化一致的熵框架下，从参考策略的补全轨迹推导过程标签，再用其训练并应用数学奖励模型。"],"why":"It changes the aggregation rule that turns verified future completions into process-label targets, linking reward construction directly to the KL-regularized optimization objective.","primary_link":"https://openreview.net/forum?id=cSxDH7N3x9","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hanningzhang/prm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HanningZhang/ER-PRM-Data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/HanningZhang/Llama3.1-Math-PRM"},{"key":"project","label":["Project","项目主页"],"url":"https://hanningzhang.github.io/math-prm/"}],"link_count":7,"sections":9},{"id":"enumerate-conjecture-prove-constructivebench-2025","title":"Enumerate-Conjecture-Prove: Formally Solving Answer-Construction Problems in Math Competitions","year":2025,"venue":"arXiv","authors":["Jialiang Sun","Yuzhi Tang","Ao Li","Chris J. Maddison","Kuldeep S. Meel"],"authors_zh":"Jialiang Sun, Yuzhi Tang, Ao Li, Chris J. Maddison, Kuldeep S. Meel","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","theorem-proving"],"tags":["lean4","answer-construction","formal-mathematics","neuro-symbolic","2025"],"status":"verified","priority":"可读","paper_type_zh":"面向数学构造题的 Lean 形式化基准与神经符号方法","best_for_zh":"需要生成并验证数学对象、而非仅完成既定定理证明的研究者。","confidence":"high","one_line":["ECP couples answer enumeration and conjecturing with Lean proofs on ConstructiveBench, turning constructed competition answers into formally checked outcomes.","ECP 在 ConstructiveBench 上把答案枚举与猜想生成连接到 Lean 证明，使竞赛构造题的候选答案可被形式化检验。"],"why":"It uses the proof kernel to check the constructed object itself, not just a textual explanation of it.","primary_link":"https://arxiv.org/abs/2505.18492","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JackSun200312/ECP"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sunjia72/ConstructiveBench"}],"link_count":5,"sections":9},{"id":"eqbench-creative-writing-v3-2025","title":"EQ-Bench Creative Writing v3","year":2025,"venue":"EQ-Bench leaderboard","authors":["EQ-Bench"],"authors_zh":"EQ-Bench","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["creative-writing","llm-judge","style-evaluation"],"tags":["benchmark","llm_judge","creative-writing"],"status":"verified","priority":"可读","paper_type_zh":"Leaderboard / LLM-judge benchmark","best_for_zh":"需要审计主观生成评测、judge model、prompt snapshot、repetition/slop 风格代理指标的读者。","confidence":"medium","one_line":["EQ-Bench Creative Writing v3 evaluates model creative-writing outputs with a judge-based leaderboard.","EQ-Bench Creative Writing v3 用 LLM judge 和风格统计列评估模型创意写作输出。"],"why":"It is a useful example of subjective generation evaluation that needs judge and prompt-snapshot audit.","primary_link":"https://eqbench.com/creative_writing.html","links":[{"key":"project","label":["Project","项目主页"],"url":"https://eqbench.com/"}],"link_count":2,"sections":9},{"id":"ernie-4-5-2025","title":"ERNIE 4.5 Technical Report","year":2025,"venue":"official technical report","authors":["Baidu-ERNIE-Team"],"authors_zh":"百度 ERNIE 团队","tracks":["frontier_reports_data_disclosure_ledger","training_usage_optimization_objectives"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr","safety_alignment"],"construction_layer":["frontier_pipeline","reward_verifier_layer"],"domains":["language","vision_language","reasoning","mathematics","code","tool_use"],"tags":["frontier-report","ernie","multimodal","sft","preference-optimization","reward-system","rlvr","disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态模型后训练与反馈系统技术报告","best_for_zh":"研究混合奖励、偏好优化、多模态 RLVR 与前沿模型披露审计的读者","confidence":"high","one_line":["ERNIE 4.5 reports a 2.3M-sample LLM SFT corpus, heterogeneous verifier/reward stack, PPO-based progressive RL with UPO, and a three-stage VLM SFT-to-RLVR pipeline, but does not release the underlying post-training ledger.","ERNIE 4.5 披露了 230 万条 LLM SFT、异构验证器/奖励栈、基于 PPO 与 UPO 的渐进 RL，以及三阶段 VLM SFT-to-RLVR 流水线，但未发布底层后训练账本。"],"why":"It is an unusually concrete frontier recipe for combining rule verification, learned judges, pairwise preference signals and multimodal environments, while showing where family-level disclosure still falls short of reproducible data lineage.","primary_link":"https://ernie.baidu.com/blog/publication/ERNIE_Technical_Report.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PaddlePaddle/ERNIE"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/baidu/ERNIE-4.5-300B-A47B-Base-Paddle"},{"key":"project","label":["Project","项目主页"],"url":"https://ernie.baidu.com/blog/posts/ernie4.5/"}],"link_count":4,"sections":9},{"id":"pathfinder-prm-error-aware-hierarchical-supervision-2025","title":"Error Typing for Smarter Rewards: Improving Process Reward Models with Error-Aware Hierarchical Supervision","year":2025,"venue":"Findings of EMNLP 2025","authors":["Tej Deep Pala","Panshul Sharma","Amir Zadeh et al."],"authors_zh":"Pala et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Error Typing for Smarter Rewards: Improving Process Reward Models with Error-Aware Hierarchical Supervision exposes process or trace supervision data.","提出 PathFinder-PRM：先区分数学错误与一致性错误，再估计步骤正确性，并用三维标签构建 40 万条分层过程监督数据。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2505.19706","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/declare-lab/PathFinder-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/declare-lab/PathFinder-600K"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/declare-lab/PathFinder-PRM-7B"}],"link_count":4,"sections":9},{"id":"evaluating-mathematical-reasoning-beyond-accuracy-2025","title":"Evaluating Mathematical Reasoning Beyond Accuracy","year":2025,"venue":"AAAI 2025 (Oral)","authors":["Shijie Xia","Xuefeng Li","Yixin Liu","Tongshuang Wu","Pengfei Liu"],"authors_zh":"Shijie Xia, Xuefeng Li, Yixin Liu, Tongshuang Wu, Pengfei Liu","tracks":["preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-evaluation","data-selection"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"数学推理过程质量评测、标注数据与评估器论文","best_for_zh":"需要在最终答案之外诊断数学推理步骤的有效性与冗余性，或筛选高质量链式推理数据的研究者。","confidence":"high","one_line":["ReasonEval evaluates mathematical reasoning beyond final-answer accuracy with step-level validity and redundancy labels, models, and selection tools.","3,000条 MR-GSM8K 解答及逐步有效性/冗余反馈，能筛选高质量推理轨迹并训练过程评估器。"],"why":"It gives a reusable process-quality target for identifying incorrect or unnecessary steps that final-answer metrics conceal.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/34987","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/ReasonEval"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/GAIR/ReasonEval-7B"}],"link_count":5,"sections":9},{"id":"evaluating-step-by-step-reasoning-traces-2025","title":"Evaluating Step-by-step Reasoning Traces: A Survey","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Jinu Lee","Julia Hockenmaier"],"authors_zh":"Jinu Lee、Julia Hockenmaier","tracks":["foundations_and_primers"],"source_role":["survey_background","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["evaluation","process_supervision"],"construction_layer":["release_audit"],"domains":["reasoning-traces","process-evaluation","evaluation-methodology"],"tags":["reasoning-trace-evaluation","process-supervision","taxonomy","EMNLP-2025"],"status":"verified","priority":"必读","paper_type_zh":"推理轨迹评测综述","best_for_zh":"关于思维链、过程标签与推理轨迹评测准则的综述。","confidence":"high","one_line":["A survey that turns “is this chain of thought good?” into four auditable evaluation dimensions.","将“推理链是否好”拆为事实性、有效性、连贯性与效用四个可审计维度的综述。"],"why":"It prevents a correct final answer from being mistaken for a validated reasoning trace.","primary_link":"https://aclanthology.org/2025.findings-emnlp.94/","links":[],"link_count":3,"sections":9},{"id":"legal-r1-rejection-distillation-2025","title":"Evaluating Test-Time Scaling LLMs for Legal Reasoning: OpenAI o1, DeepSeek-R1, and Beyond","year":2025,"venue":"Findings of EMNLP 2025","authors":["Yinghao Hu","Yaoyao Yu","Leilei Gan","Bin Wei","Kun Kuang","Fei Wu"],"authors_zh":"Yinghao Hu、Yaoyao Yu、Leilei Gan、Bin Wei、Kun Kuang、Fei Wu","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["legal_reasoning","chinese_law","us_law"],"tags":["legal-r1","rejection-distillation","legal-reasoning","deepseek-r1","progressive-sft","legalbench","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"法律推理拒绝蒸馏与发布审计","best_for_zh":"研究高风险领域推理数据、答案验证和双语法律模型的读者","confidence":"high","one_line":["Legal-R1 retains legal reasoning traces whose answers match gold labels within up to three DeepSeek-R1 attempts, reporting 96,533 bilingual SFT records while the verified official repository remains empty.","Legal-R1在最多三次DeepSeek-R1尝试中保留答案匹配金标准的法律推理轨迹，报告形成96,533条中美法律SFT数据，但官方仓库核验时为空。"],"why":"The paper makes its rejection-distillation and staged-training logic inspectable, while exposing the answer-level verifier, lineage, legal-validity, licensing, and release gaps that must be resolved before the data can be audited or reused.","primary_link":"https://aclanthology.org/2025.findings-emnlp.742/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/YinghaoHu/Legal-R1-14B"}],"link_count":5,"sections":9},{"id":"every-rollout-counts-2025","title":"Every Rollout Counts: Optimal Resource Allocation for Efficient Test-Time Scaling","year":2025,"venue":"NeurIPS 2025","authors":["Xinglin Wang","Yiwei Li","Shaoxiong Feng","Peiwen Yuan","Yueqi Zhang","Jiayi Shi","Chuyi Tan","Boyuan Pan","Yao Hu","Kan Li"],"authors_zh":"unknown","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematics","test_time_search","reasoning"],"tags":["track5","online_ttc_trace_reuse"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["DORA uses PRM quality and embedding-derived semantic uniqueness to allocate a fixed rollout budget across soft reasoning directions during parallel test-time search.","DORA 将固定 rollout 预算分配到语义推理方向而不是孤立候选，以降低冗余采样。"],"why":"For rollout_search_test_time_trace_data, DORA specifies which per-step scores, embeddings, affinities, allocations, continuations, and final votes must be logged to audit how test-time trace reuse converts a fixed budget into search outcomes.","primary_link":"https://arxiv.org/abs/2506.15707","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WangXinglin/DORA"}],"link_count":5,"sections":9},{"id":"epo-group-preference-estimation-2025","title":"Expectation Preference Optimization: Reliable Preference Estimation for Improving the Reasoning Capability of Large Language Models","year":2025,"venue":"EMNLP 2025 Main Conference","authors":["Zelin Li","Dawei Song"],"authors_zh":"Zelin Li, Dawei Song","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["preference-optimization","group-preference","self-training","reasoning","verifier"],"status":"verified","priority":"必读","paper_type_zh":"面向推理自训练的组级偏好估计研究","best_for_zh":"从每个提示的多次可验证推理采样构造偏好数据的读者。","confidence":"high","one_line":["EPO learns from groups of correct and incorrect sampled responses, estimating a group expectation rather than trusting one preference pair.","EPO 从成组的正确与错误采样回答中学习，以组期望估计取代对单个偏好对的信任。"],"why":"It makes multi-sample grouping and within-group probability weighting part of the training signal rather than using one arbitrary response pair.","primary_link":"https://aclanthology.org/2025.emnlp-main.1532/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Vespertinus9/EPO"}],"link_count":3,"sections":9},{"id":"expertlongbench-long-form-2025","title":"ExpertLongBench: Benchmarking Language Models on Expert-Level Long-Form Generation Tasks with Structured Checklists","year":2025,"venue":"ICLR 2026","authors":["Jie Ruan","Inderjeet Nair","Shuyang Cao","Amy Liu","Sheza Munir","Micah Pollens-Dempsey","Tiffany Chiang","Lucy Kates","Nicholas David","Sihan Chen","Ruxin Yang","Yuqian Yang","Jasmine Gump","Tessa Bialek","Vivek Sankaran","Margo Schlanger","Lu Wang"],"authors_zh":"Jie Ruan, Inderjeet Nair, Shuyang Cao, Amy Liu, Sheza Munir 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["long_form_generation","professional_work","cross_domain"],"tags":["long_form","rubric","checklist"],"status":"verified","priority":"必读","paper_type_zh":"专家级长文生成的 rubric 与结构化清单评测基准","best_for_zh":"研究长文本生成、专家 Judge 与清单式评测的读者。","confidence":"high","one_line":["ExpertLongBench scores expert-level long-form outputs with task rubrics and grounded checklist comparisons.","ExpertLongBench 用专家 rubric 与 CLEAR 清单框架核验长篇专业生成内容是否正确满足要求。"],"why":"It separates generating required aspects from generating them correctly in long professional outputs.","primary_link":"https://arxiv.org/abs/2506.01241","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/launch/ExpertLongBench"}],"link_count":4,"sections":9},{"id":"explorer-web-trajectories-2025","title":"Explorer: Scaling Exploration-driven Web Trajectory Synthesis for Multimodal Web Agents","year":2025,"venue":"Findings of ACL 2025","authors":["Vardaan Pahuja","Yadong Lu","Corby Rosset","Boyu Gou","Arindam Mitra","Spencer Whitehead","Yu Su","Ahmed Hassan Awadallah"],"authors_zh":"Vardaan Pahuja、Yadong Lu、Corby Rosset、Boyu Gou、Arindam Mitra、Spencer Whitehead、Yu Su、Ahmed Hassan Awadallah","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","agent_environment","scaling_study"],"verification_contract":["judgment_required","environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","release_audit"],"domains":["web_navigation","gui_agents","information_seeking","transactional_web_tasks"],"tags":["explorer","web-agent","gui-agent","trajectory-synthesis","live-web","playwright","successful-only-filter","multimodal-sft"],"status":"partial","priority":"可读","paper_type_zh":"多模态网页轨迹合成与筛选配方","best_for_zh":"研究网页智能体训练、实时环境轨迹构造、学习型 verifier 与数据发布审计的读者","confidence":"high","one_line":["Synthesizes 175K GPT-4o-driven live-web episodes and retains 94K verifier-approved multimodal trajectories for web-agent training.","Explorer 从 175K 次实时网页尝试中保留 94K 条 GPT-4o verifier 接受的多模态轨迹，用于网页智能体 SFT；但官方未提供该语料的不可变快照、清单或专属许可。"],"why":"Shows how exploration, environmental execution, multimodal state capture, and outcome filtering can scale agent data, while exposing why immutable snapshots and content rights are essential release metadata.","primary_link":"https://aclanthology.org/2025.findings-acl.326/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OSU-NLP-Group/Explorer"},{"key":"project","label":["Project","项目主页"],"url":"https://osu-nlp-group.github.io/Explorer/"}],"link_count":6,"sections":9},{"id":"factory-long-form-factuality-2025","title":"FACTORY: A Challenging Human-Verified Prompt Set for Long-Form Factuality","year":2025,"venue":"ICLR 2026","authors":["Mingda Chen","Yang Li","Xilun Chen","Adina Williams","Gargi Ghosh","Scott Yih"],"authors_zh":"Mingda Chen, Yang Li, Xilun Chen, Adina Williams, Gargi Ghosh, Scott Yih","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["factuality","long_form_generation"],"tags":["factuality","long_form","human_verification","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"人工核验的长文本事实性提示与证据评测集","best_for_zh":"需要构建或审计长文本事实性 Judge、证据检索和 claim 级评测的研究者。","confidence":"high","one_line":["FACTORY pairs difficult human-refined long-form prompts with claim-level evidence annotations.","FACTORY 将人工修订的困难长文本提示与 claim 级证据标注配对，用于可靠评估模型事实性。"],"why":"It separates model factuality failures from defects in automatically generated prompts.","primary_link":"https://arxiv.org/abs/2508.00109","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/FACTORY"}],"link_count":4,"sections":9},{"id":"faithbench-a-diverse-hallucination-benchmark-for-summarization-by-modern-llms-2025","title":"FaithBench: A Diverse Hallucination Benchmark for Summarization by Modern LLMs","year":2025,"venue":"NAACL 2025 (Short Papers)","authors":["Forrest Sheng Bao","Miaoran Li","Renyi Qu","Ge Luo","Erana Wan","Yujia Tang","Weisi Fan","Manveer Singh Tamber","Suleman Kazi","Vivek Sourabh","Mike Qi","RuiXuan Tu","Chenyu Xu","Matthew Gonzales","Ofer Mendelevitch","Amin Ahmad"],"authors_zh":"Forrest Sheng Bao、Miaoran Li、Renyi Qu、Ge Luo、Erana Wan、Yujia Tang、Weisi Fan、Manveer Singh Tamber、Suleman Kazi、Vivek Sourabh、Mike Qi、RuiXuan Tu、Chenyu Xu、Matthew Gonzales、Ofer Mendelevitch、Amin Ahmad","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🧭 Surveys and Primers map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","以多样化幻觉场景检查摘要忠实性，提醒读者不能只看流畅度或单一自动指标。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://aclanthology.org/2025.naacl-short.38/","links":[],"link_count":3,"sections":9},{"id":"latency-aware-tts-2025","title":"Faster and Better LLMs via Latency-Aware Test-Time Scaling","year":2025,"venue":"Findings of EMNLP 2025","authors":["Zili Wang","Tianyu Zhang","Haoli Bai","Lu Hou","Xianzhi Yu","Wulong Liu","Shiming Xiang","Lei Zhu"],"authors_zh":"Zili Wang、Tianyu Zhang、Haoli Bai、Lu Hou、Xianzhi Yu、Wulong Liu、Shiming Xiang、Lei Zhu（机构：中国科学院大学、中国科学院自动化研究所、华为诺亚方舟实验室）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-compute","latency","concurrency","speculative-decoding","parallel-scaling"],"status":"verified","priority":"必读","paper_type_zh":"面向系统部署的测试时扩展研究（Findings of EMNLP 2025）","best_for_zh":"在服务级延迟约束下部署推理模型的读者。","confidence":"high","one_line":["Latency-aware TTS allocates parallel branches and speculative decoding capacity to maximize reasoning accuracy under a time limit.","延迟感知测试时扩展在固定时间内分配并行分支与推测解码能力，以提高推理准确率。"],"why":"It shows that token-optimal inference strategies can be slower than concurrent alternatives at the same accuracy.","primary_link":"https://aclanthology.org/2025.findings-emnlp.928/","links":[],"link_count":2,"sections":9},{"id":"fastmcts-2025","title":"FastMCTS: A Simple Sampling Strategy for Data Synthesis","year":2025,"venue":"ACL 2025","authors":["Peiji Li","Kai Lv","Yunfan Shao","Yichuan Ma","Linyang Li","Xiaoqing Zheng","Xipeng Qiu","Qipeng Guo"],"authors_zh":"Peiji Li、Kai Lv、Yunfan Shao、Yichuan Ma、Linyang Li、Xiaoqing Zheng、Xipeng Qiu、Qipeng Guo","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","process_supervision","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","state_action_level","pairwise_preference","trajectory_value"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics","english_math","chinese_math"],"tags":["synthetic-reasoning-data","monte-carlo-tree-search","tree-structured-data","rejection-sampling","step-level-preference","branch-dpo","mathematical-reasoning","llm-verifier","token-budget","code-release"],"status":"partial","priority":"必读","paper_type_zh":"树搜索数据合成与过程监督配方","best_for_zh":"研究推理数据合成、搜索树轨迹、终端验证、步骤偏好与 token 预算审计的读者","confidence":"high","one_line":["Reuses reasoning prefixes in an MCTS-inspired tree and backs up terminal LLM judgments to construct math SFT paths and step/branch preference pairs under a token budget.","在 MCTS 风格的共享树中复用推理前缀，并把终端 LLM 判断回传为节点值，从而构造数学 SFT 路径与步骤/分支偏好对。"],"why":"Connects search allocation, terminal verification, tree-derived process labels, SFT, and DPO while making the label assumptions and missing experimental data auditable.","primary_link":"https://aclanthology.org/2025.acl-long.1190/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/FlyingDutchman26/FastMCTS"}],"link_count":4,"sections":9},{"id":"fb-bench-human-feedback-2025","title":"FB-Bench: A Fine-Grained Multi-Task Benchmark for Evaluating LLMs’ Responsiveness to Human Feedback","year":2025,"venue":"EMNLP 2025","authors":["Youquan Li","Miao Zheng","Fan Yang","Guosheng Dong","Bin Cui","Weipeng Chen","Zenan Zhou","Wentao Zhang"],"authors_zh":"Youquan Li、Miao Zheng、Fan Yang、Guosheng Dong、Bin Cui、Weipeng Chen、Zenan Zhou、Wentao Zhang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","answer_level"],"training_use":["evaluation","audit"],"construction_layer":["filtering_curation","reward_verifier_layer","release_audit"],"domains":["human_feedback","chinese","dialogue"],"tags":["human_feedback","checklist","llm_judge","chinese"],"status":"verified","priority":"可读","paper_type_zh":"人类反馈基准论文","best_for_zh":"研究多轮对话、反馈利用与 checklist 型 Judge 的研究者。","confidence":"high","one_line":["FB-Bench tests whether an LLM can use Chinese user feedback to repair a deficient prior response, rather than merely answer a new turn.","FB-Bench 用任务、原始缺陷回答与用户反馈组成 591 个中文样本，评测模型能否真正按反馈修正回答。"],"why":"It isolates an operational feedback contract that ordinary single-turn benchmarks miss.","primary_link":"https://aclanthology.org/2025.emnlp-main.471/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PKU-Baichuan-MLSystemLab/FB-Bench"}],"link_count":4,"sections":9},{"id":"fea-bench-feature-implementation-2025","title":"FEA-Bench: A Benchmark for Evaluating Repository-Level Code Generation for Feature Implementation","year":2025,"venue":"ACL 2025","authors":["Wei Li","Xin Zhang","Zhongxin Guo","Shaoguang Mao","Wen Luo","Guangyue Peng","Yangyu Huang","Houfeng Wang","Scarlett Li"],"authors_zh":"Wei Li, Xin Zhang, Zhongxin Guo, Shaoguang Mao, Wen Luo, Guangyue Peng, Yangyu Huang, Houfeng Wang, Scarlett Li","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","unit-tests"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"仓库级功能实现代码生成基准","best_for_zh":"需要评估新增功能实现而非缺陷修复的仓库级代码智能体的研究者。","confidence":"high","one_line":["FEA-Bench derives 1,401 repository-level feature-implementation tasks from pull requests and validates solutions with associated unit tests.","FEA-Bench 从 83 个仓库的拉取请求提取 1,401 个功能开发任务，并用关联单元测试验证仓库级新增功能。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://aclanthology.org/2025.acl-long.839/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/FEA-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/microsoft/FEA-Bench"}],"link_count":3,"sections":9},{"id":"feat-preference-feedback-tutoring-2025","title":"FEAT: A Preference Feedback Dataset through a Cost-Effective Auto-Generation and Labeling Framework for English AI Tutoring","year":2025,"venue":"ACL 2025","authors":["Hyein Seo","Taewook Hwang","Yohan Lee","Sangkeun Jung"],"authors_zh":"Hyein Seo、Taewook Hwang、Yohan Lee、Sangkeun Jung","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈数据审计","confidence":"","one_line":["FEAT releases tutoring prompts with candidate instructional answers, textual feedback, and human ranking signals for bounded preference learning, reward modeling, evaluation, and audit.","FEAT 发布围绕其任务场景组织的候选回答比较与反馈记录，可用于有边界的偏好学习、奖励建模、评测和审计。"],"why":"The release supports feedback learning for 英语辅导中的教师反馈质量.","primary_link":"https://aclanthology.org/2025.acl-short.45/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hyenee/FEAT"}],"link_count":2,"sections":9},{"id":"sweet-spot-preference-construction-2025","title":"Finding the Sweet Spot: Preference Data Construction for Scaling Preference Optimization","year":2025,"venue":"ACL 2025 Long Papers","authors":["Yao Xiao","Hai Ye","Linyao Chen","Hwee Tou Ng","Lidong Bing","Xiaoli Li","Roy Ka-Wei Lee"],"authors_zh":"Yao Xiao, Hai Ye, Linyao Chen, Hwee Tou Ng, Lidong Bing, Xiaoli Li, Roy Ka-Wei Lee","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["instruction-tuning","alignment"],"tags":["preference-data","dpo","reward-distribution","data-construction"],"status":"verified","priority":"必读","paper_type_zh":"在线偏好数据构造研究","best_for_zh":"扩展 DPO 回答采样且不想盲目使用最大—最小对的读者。","confidence":"high","one_line":["The paper constructs DPO pairs from reward-distribution positions, preferring a μ−2σ rejected response over the minimum.","该工作按奖励分布位置构造 DPO 对，优先使用均值减二倍标准差的拒绝回答而非最小值。"],"why":"It turns a reward-ranked response pool into a calibrated pair-selection policy for the DPO training consumer.","primary_link":"https://aclanthology.org/2025.acl-long.615/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XYaoooo/DPO_Pair"}],"link_count":3,"sections":9},{"id":"wmt25-multilingual-instruction-shared-task-2025","title":"Findings of the WMT25 Multilingual Instruction Shared Task: Persistent Hurdles in Reasoning, Generation, and Evaluation","year":2025,"venue":"WMT 2025","authors":["Tom Kocmi","Ekaterina Artemova","Eleftherios Avramidis","Eleftheria Briakou","Pinzhen Chen","Marzieh Fadaee","Markus Freitag","Roman Grundkiewicz","Yupeng Hou","Philipp Koehn","Julia Kreutzer","Saab Mansour","Stefano Perrella","Lorenzo Proietti","Parker Riley","Eduardo Sánchez","Patrícia Schmidtová","Mariya Shmatova","Vilém Zouhar"],"authors_zh":"Tom Kocmi、Ekaterina Artemova、Eleftherios Avramidis、Eleftheria Briakou、Pinzhen Chen、Marzieh Fadaee、Markus Freitag、Roman Grundkiewicz、Yupeng Hou、Philipp Koehn、Julia Kreutzer、Saab Mansour、Stefano Perrella、Lorenzo Proietti、Parker Riley、Eduardo Sánchez、Patrícia Schmidtová、Mariya Shmatova、Vilém Zouhar","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["WMT25 MIST releases 30-language reasoning, generation, translation, summarization, and LLM-judge tasks with human evaluation.","WMT25 MIST 跨 30 种语言发布推理、生成、翻译、摘要与 LLM 裁判任务及人类评测。"],"why":"WMT25 MIST releases 30-language reasoning, generation, translation, summarization, and LLM-judge tasks with human evaluation.","primary_link":"https://aclanthology.org/2025.wmt-1.23/","links":[{"key":"data","label":["Data","数据"],"url":"https://github.com/wmt-conference/wmt25-mist"}],"link_count":2,"sections":9},{"id":"fixing-distribution-shifts-of-llm-self-critique-2025","title":"Fixing Distribution Shifts of LLM Self-Critique via On-Policy Self-Play Training","year":2025,"venue":"ACL 2025","authors":["Rong Bao","Donglei Yu","Kai Fan","Minpeng Liao"],"authors_zh":"Rong Bao、Donglei Yu、Kai Fan、Minpeng Liao","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward","process_reward"],"training_use":["sft","process_supervision","rlvr"],"construction_layer":["trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["math","self-critique"],"tags":["on-policy","self-play","self-critique","correction","monte-carlo-reward","process-reward","ppo","distribution-shift"],"status":"partial","priority":"可读","paper_type_zh":"On-policy 自我批评与纠正训练配方","best_for_zh":"研究批评数据漂移、过程奖励、self-play buffer 与数学 RLVR 的读者","confidence":"medium","one_line":["SCOP refreshes reasoning-critique-correction episodes from the current policy and scores critiques through sampled repair success, but its official repository remains an implementation-free placeholder.","SCOP 从当前策略刷新推理-批评-纠正 episode，并按采样修复成功率奖励批评，但官方仓库仍无实现与轨迹发布。"],"why":"It makes critique-data freshness and stage-specific feedback explicit construction variables while exposing how teacher initialization, model-relative prompt filtering, Monte Carlo noise, and missing trajectory lineage limit reuse.","primary_link":"https://aclanthology.org/2025.acl-long.865/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rbao2018/SCOP"}],"link_count":4,"sections":9},{"id":"flames-2025","title":"FLAMES: Improving LLM Math Reasoning via a Fine-Grained Analysis of the Data Synthesis Pipeline","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Parker Seegmiller","Kartik Mehta","Soumya Saha","Chenyang Tao","Shereen Oraby","Arpit Gupta","Tagyoung Chung","Mohit Bansal","Nanyun Peng"],"authors_zh":"Parker Seegmiller、Kartik Mehta、Soumya Saha、Chenyang Tao、Shereen Oraby、Arpit Gupta、Tagyoung Chung、Mohit Bansal、Nanyun Peng","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline"],"domains":["mathematics"],"tags":["synthetic-math-data","problem-synthesis","data-filtering","self-consistency","data-mixture","teacher-distillation"],"status":"partial","priority":"必读","paper_type_zh":"数学合成数据管线消融与扩展研究","best_for_zh":"研究合成数学数据、教师蒸馏、质量过滤、数据混合与覆盖率权衡的读者","confidence":"high","one_line":["FLAMES compares 12 math-problem agents and six quality-control strategies, then defines 150K, 1M, and 1.5M mixtures that favor coverage and the first teacher solution; the data and construction code are not officially released.","FLAMES 在统一教师、学生与训练设置下比较 12 种数学题合成 agent 和 6 种质量控制策略，并给出 150K、1M、1.5M 混合配方；但最终采用 First 且无独立解答验证，官方数据与构造代码也未发布。"],"why":"It exposes a concrete coverage-versus-verification trade-off, exact mixture recipe, teacher sensitivity, lexical decontamination rule, and a failure case where an LLM solvability filter rejects valid hard problems.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1346/","links":[],"link_count":4,"sections":9},{"id":"fmc-olympiad-lean-autoformalization-2025","title":"FMC: Formalization of Natural Language Mathematical Competition Problems","year":2025,"venue":"AI for Math Workshop at ICML 2025","authors":["Jiaxuan Xie","Chengwu Liu","Ye Yuan","Siqi Li","Zhiping Xiao","Ming Zhang"],"authors_zh":"Jiaxuan Xie, Chengwu Liu, Ye Yuan, Siqi Li, Zhiping Xiao, Ming Zhang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","autoformalization"],"tags":["autoformalization","lean4","olympiad-math","formal-verification","2025"],"status":"verified","priority":"可读","paper_type_zh":"奥数自然语言题到 Lean 的自动形式化数据集","best_for_zh":"需要自然语言到 Lean 形式化、自动定理证明训练或评测数据的研究者。","confidence":"high","one_line":["FMC aligns 3,922 Olympiad problems with 9,787 Lean formalizations using an error-feedback pipeline and Lean validation.","FMC 通过带错误反馈的自动形式化流程，将 3,922 道奥数自然语言题与 9,787 个 Lean 形式化对齐并验证。"],"why":"It exposes Lean validation within a large-scale natural-language formalization pipeline.","primary_link":"https://arxiv.org/abs/2507.11275","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JadeXie1205/FMC"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/JadeXie1205/FMC"}],"link_count":6,"sections":9},{"id":"focalpo-correct-preference-ranking-2025","title":"FocalPO: Enhancing Preference Optimizing by Focusing on Correct Preference Rankings","year":2025,"venue":"ACL 2025 Short Papers","authors":["Tong Liu","Xiao Yu","Wenxuan Zhou","Jindong Gu","Volker Tresp"],"authors_zh":"Tong Liu, Xiao Yu, Wenxuan Zhou, Jindong Gu, Volker Tresp（慕尼黑大学、哥伦比亚大学、南加州大学、牛津大学、慕尼黑机器学习中心）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","sample-weighting","focal-loss","dpo"],"status":"verified","priority":"必读","paper_type_zh":"偏好回答对加权与优化目标研究","best_for_zh":"适合决定当前排序错误的困难偏好回答对是否应主导离线对齐更新的读者。","confidence":"high","one_line":["FocalPO reweights DPO so preference pairs with correct current implicit rankings contribute more than misranked pairs.","FocalPO 重加权 DPO，使当前隐式排序正确的偏好回答对比排序错误的回答对贡献更大。"],"why":"It challenges DPO's default emphasis on pairs the model currently gets wrong and makes ranking state an explicit training-use signal.","primary_link":"https://aclanthology.org/2025.acl-short.21/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TongLiu-github/focalpo"}],"link_count":4,"sections":9},{"id":"focused-dpo-error-prone-code-2025","title":"Focused-DPO: Enhancing Code Generation Through Focused Preference Optimization on Error-Prone Points","year":2025,"venue":"Findings of ACL 2025","authors":["Kechi Zhang","Ge Li","Jia Li","Yihong Dong","Jia Li","Zhi Jin"],"authors_zh":"Kechi Zhang, Ge Li, Jia Li, Yihong Dong, Jia Li, Zhi Jin（北京大学、清华大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["code_generation","preference_learning"],"tags":["code-generation","preference-data","error-localization","dpo"],"status":"verified","priority":"必读","paper_type_zh":"可执行偏好数据构造与细粒度优化研究","best_for_zh":"适合训练代码模型、且少量局部选择即可决定程序能否通过测试的读者。","confidence":"high","one_line":["Focused-DPO identifies error-prone code segments from executable preference pairs and gives those segments greater DPO influence.","Focused-DPO 从可执行偏好回答对中定位易错代码片段，并让这些片段在 DPO 中拥有更大影响。"],"why":"It turns test-sensitive code differences into an explicit record field and optimization weight instead of averaging them with routine syntax.","primary_link":"https://aclanthology.org/2025.findings-acl.498/","links":[],"link_count":3,"sections":9},{"id":"formalmath-formal-mathematical-reasoning-2025","title":"FormalMATH: Benchmarking Formal Mathematical Reasoning of Large Language Models","year":2025,"venue":"arXiv","authors":["Zhouliang Yu","Ruotian Peng","Keyi Ding","Yizhe Li","Zhongyuan Peng","Minghao Liu","Yifan Zhang","Zheng Yuan","Huajian Xin","Wenhao Huang","Yandong Wen","Ge Zhang","Weiyang Liu"],"authors_zh":"Zhouliang Yu, Ruotian Peng, Keyi Ding, Yizhe Li, Zhongyuan Peng, Minghao Liu, Yifan Zhang, Zheng Yuan, Huajian Xin, Wenhao Huang, Yandong Wen, Ge Zhang, Weiyang Liu","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","theorem-proving"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"大规模 Lean4 形式数学推理基准","best_for_zh":"需要评测 Lean4 定理证明、形式化陈述或可验证数学训练数据的研究者。","confidence":"high","one_line":["FormalMATH releases 5,560 Lean4-verified mathematical problems and an autoformalization pipeline for broad formal-reasoning evaluation.","FormalMATH 以 5,560 道经 Lean4 验证的问题覆盖从奥赛到本科数学，并提供自动形式化与语义筛选流程。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://arxiv.org/abs/2505.02735","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Sphere-AI-Lab/FormalMATH-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SphereLab/FormalMATH-All"},{"key":"project","label":["Project","项目主页"],"url":"https://sphere-ai-lab.github.io/FormalMATH/"}],"link_count":4,"sections":9},{"id":"freeprm-implicit-process-rewards-2025","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","year":2025,"venue":"arXiv preprint under review","authors":["Lin Sun","Chuang Liu","Xiaofeng Ma","Tao Yang","Weijia Lu","Ning Wu"],"authors_zh":"Lin Sun、Chuang Liu、Xiaofeng Ma、Tao Yang、Weijia Lu、Ning Wu（UAES AI Lab）","tracks":["training_usage_optimization_objectives"],"source_role":["process_supervision","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["process_reward"],"training_use":["process_supervision","test_time_compute"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning"],"tags":["process-reward-model","weak-supervision","pseudo-labels","buffer-probability"],"status":"verified","priority":"必读","paper_type_zh":"弱监督过程标签构造与过程奖励模型训练研究","best_for_zh":"适合在只能廉价获得最终答案正确性时构建过程奖励监督的读者。","confidence":"medium","one_line":["FreePRM learns step rewards from final-answer labels alone by adding a buffer state that absorbs uncertainty in trajectory-level pseudo supervision.","FreePRM 仅从最终答案标签学习步骤奖励，并用缓冲状态吸收轨迹级伪监督中的不确定性。"],"why":"It turns the gap between outcome labels and step labels into an explicit uncertainty-aware training objective.","primary_link":"https://arxiv.org/abs/2506.03570","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sunlin-ai/FreePRM"}],"link_count":3,"sections":9},{"id":"scientific-discovery-llm-survey-2025","title":"From Automation to Autonomy: A Survey on Large Language Models in Scientific Discovery","year":2025,"venue":"EMNLP 2025","authors":["Tianshi Zheng","Zheye Deng","Hong Ting Tsang","Weiqi Wang","Jiaxin Bai","Zihao Wang","Yangqiu Song"],"authors_zh":"Tianshi Zheng 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["scientific-discovery","agents","reasoning"],"tags":["foundations-and-primers","scientific-discovery","agents","emnlp-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"科学发现中的大语言模型综述","best_for_zh":"关注智能体、科学工作流和人机协作的读者。","confidence":"high","one_line":["An EMNLP 2025 survey of how LLM roles in scientific discovery evolve from tools to increasingly autonomous agents.","综述大语言模型在科学发现中从自动化工具走向更自主角色的变化。"],"why":"It gives readers a concrete vocabulary for separating task automation from broader scientific responsibility.","primary_link":"https://aclanthology.org/2025.emnlp-main.895/","links":[],"link_count":2,"sections":9},{"id":"generation-to-judgment-survey-2025","title":"From Generation to Judgment: Opportunities and Challenges of LLM-as-a-Judge","year":2025,"venue":"EMNLP 2025","authors":["Dawei Li","Bohan Jiang","Liangjie Huang","Alimohammad Beigi","Chengshuai Zhao","Zhen Tan","Amrita Bhattacharjee","Yuxuan Jiang","Canyu Chen","Tianhao Wu","Kai Shu","Lu Cheng","Huan Liu"],"authors_zh":"Dawei Li, Bohan Jiang, Liangjie Huang, Alimohammad Beigi, Chengshuai Zhao, Zhen Tan, Amrita Bhattacharjee, Yuxuan Jiang, Canyu Chen, Tianhao Wu, Kai Shu, Lu Cheng, Huan Liu","tracks":["audit_failure_contamination_verifier_attacks","foundations_and_primers"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Top-conference survey mapping judge inputs, outputs, benchmarks, and reliability risks.","以输入输出、方法、基准三维分类，系统梳理 LLM judge 的用途与可靠性风险。"],"why":"It maps where LLM judges are used and which biases, vulnerabilities, and benchmarking gaps constrain their reliability.","primary_link":"https://aclanthology.org/2025.emnlp-main.138/","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/2025.emnlp-main.138.pdf"}],"link_count":4,"sections":9},{"id":"hypothesis-discovery-rule-learning-survey-2025","title":"From Reasoning to Learning: A Survey on Hypothesis Discovery and Rule Learning with Large Language Models","year":2025,"venue":"TMLR 2025","authors":["Kaiyu He","Zhiyu Chen"],"authors_zh":"Kaiyu He、Zhiyu Chen","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["reward_verifier_layer","trace_writing"],"domains":["reasoning","hypothesis-discovery","rule-learning","evaluation"],"tags":["foundations-and-primers","reasoning","hypothesis-discovery","rule-learning","tmlr-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"假设发现与规则学习综述","best_for_zh":"研究科学发现、规则归纳或语言模型生成假设评测的读者。","confidence":"high","one_line":["A TMLR 2025 survey that frames LLM hypothesis discovery as generation, application, and validation.","以溯因、演绎、归纳为框架，梳理语言模型生成、应用和验证假设的研究。"],"why":"It makes the difference clear between producing an explanation and testing whether that explanation learns something reliable.","primary_link":"https://openreview.net/forum?id=d7W38UzUg0","links":[],"link_count":3,"sections":9},{"id":"survey-of-reasoning-large-language-models-2025","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","year":2025,"venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 2025","authors":["Zhong-Zhi Li","Duzhen Zhang","Ming-Liang Zhang","Jiaxin Zhang","Zengyan Liu","Yuxuan Yao","Haotian Xu","Junhao Zheng","Pei-Jie Wang","Xiuyi Chen","Yingying Zhang","Fei Yin","Jiahua Dong","Zhiwei Li","Bao-Long Bi","Ling-Rui Mei","Junfeng Fang","Xiao Liang","Zhijiang Guo","Le Song","Cheng-Lin Liu"],"authors_zh":"Zhong-Zhi Li, Duzhen Zhang, Ming-Liang Zhang, Jiaxin Zhang, Zengyan Liu, Yuxuan Yao, Haotian Xu, Junhao Zheng, Pei-Jie Wang, Xiuyi Chen, Yingying Zhang, Fei Yin, Jiahua Dong, Zhiwei Li, Bao-Long Bi, Ling-Rui Mei, Junfeng Fang, Xiao Liang, Zhijiang Guo, Le Song, Cheng-Lin Liu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["reasoning","survey"],"tags":["survey","reasoning-llms","foundation-models"],"status":"verified","priority":"必读","paper_type_zh":"推理大模型综述与资源索引","best_for_zh":"需要选择推理数据配方、验证机制或评测基准的研究者。","confidence":"high","one_line":["Surveys the transition from fast System 1-style LLM behavior to deliberate reasoning LLMs, organizing construction methods, benchmarks, metrics, and open challenges.","以 System 1 到 System 2 为主线，梳理推理大模型的构建方法、基准、指标与开放问题。"],"why":"It supplies a cross-method map before readers choose a reasoning-data recipe, a verifier, or an evaluation benchmark.","primary_link":"https://doi.org/10.1109/TPAMI.2025.3637037","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zzli2022/Awesome-System2-Reasoning-LLM"}],"link_count":3,"sections":9},{"id":"gemini-3-pro-fsf-report-2025","title":"Frontier Safety Framework Report - Gemini 3 Pro (November, 2025) v2","year":2025,"venue":"Google DeepMind Frontier Safety Framework report","authors":["Google DeepMind"],"authors_zh":"Google DeepMind","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","safety_alignment","audit"],"construction_layer":["trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["reasoning","agentic_systems","cybersecurity","safety"],"tags":["gemini-3-pro","gemini","google-deepmind","frontier-report","frontier-safety-framework","thought-traces","sft","reinforcement-learning","safety-audit","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿安全框架报告与数据披露台账","best_for_zh":"需要审查前沿推理模型的后训练数据对象、奖励边界、智能体评测脚手架与可审计性缺口的读者","confidence":"high","one_line":["Google DeepMind's Gemini 3 Pro FSF report discloses model-generated thought traces for SFT, RL length penalties on thoughts, high-level guardrails, and concrete evaluation scaffolding, but not the underlying prompts, traces, rewards, or reproducibility artifacts.","Google DeepMind 的 Gemini 3 Pro FSF 报告披露了用于 SFT 的推理模型生成 thought traces、施加于 thought 的 RL 长度惩罚、高层 guardrails 和具体评测脚手架，但未公开底层 prompts、traces、奖励或可复现制品。"],"why":"It makes a closed frontier post-training interface partially auditable: the report separates model-generated thought-trace SFT from reward handling and exposes evaluation-budget confounders, while making clear why data, reward, and audit reuse remains impossible.","primary_link":"https://storage.googleapis.com/deepmind-media/gemini/gemini_3_pro_fsf_report.pdf","links":[],"link_count":2,"sections":9},{"id":"frontiercs-evolving-code-reasoning-2025","title":"FrontierCS: Evolving Challenges for Evolving Intelligence","year":2025,"venue":"arXiv preprint","authors":["Qiuyang Mang","Wenhao Chai","Zhifei Li","Huanzhi Mao","Shang Zhou","Alexander Du","Hanchen Li","Shu Liu","Edwin Chen","Yichuan Wang","Xieting Chu","Zerui Cheng","Yuan Xu","Tian Xia","Zirui Wang","Tianneng Shi","Jianzhu Yao","Yilong Zhao","Qizheng Zhang","Charlie Ruan","Zeyu Shen","Kaiyuan Liu","Runyuan He","Dong Xing","Zerui Li","Zirong Zeng","Yige Jiang","Lufeng Cheng","Ziyi Zhao","Youran Sun","Wesley Zheng","Meiyuwang Zhang","Ruyi Ji","Xuechang Tu","Zihan Zheng","Zexing Chen","Kangyang Zhou","Zhaozi Wang","Jingbang Chen","Aleksandra Korolova","Peter Henderson","Pramod Viswanath","Vijay Ganesh","Saining Xie","Zhuang Liu","Dawn Song","Sewon Min","Ion Stoica"],"authors_zh":"Qiuyang Mang、Wenhao Chai、Zhifei Li、Huanzhi Mao、Shang Zhou、Alexander Du、Hanchen Li、Shu Liu、Edwin Chen、Yichuan Wang、Xieting Chu、Zerui Cheng、Yuan Xu、Tian Xia、Zirui Wang、Tianneng Shi、Jianzhu Yao、Yilong Zhao、Qizheng Zhang、Charlie Ruan、Zeyu Shen、Kaiyuan Liu、Runyuan He、Dong Xing、Zerui Li、Zirong Zeng、Yige Jiang、Lufeng Cheng、Ziyi Zhao、Youran Sun、Wesley Zheng、Meiyuwang Zhang、Ruyi Ji、Xuechang Tu、Zihan Zheng、Zexing Chen、Kangyang Zhou、Zhaozi Wang、Jingbang Chen、Aleksandra Korolova、Peter Henderson、Pramod Viswanath、Vijay Ganesh、Saining Xie、Zhuang Liu、Dawn Song、Sewon Min、Ion Stoica","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"可读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["FrontierCS tracks evolving model limits with expert-designed open computer-science problems and partial scoring.","用专家设计、可部分评分的开放计算机科学题，持续追踪前沿模型的真实能力边界。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://arxiv.org/abs/2512.15699","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/FrontierCS/Frontier-CS"},{"key":"project","label":["Project","项目主页"],"url":"https://www.frontier-cs.org/"}],"link_count":3,"sections":9},{"id":"full-step-dpo-stepwise-rewards-2025","title":"Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning","year":2025,"venue":"Findings of ACL 2025","authors":["Huimin Xu","Xin Mao","Feng-Lin Li","Xiaobao Wu","Wang Chen","Wei Zhang","Anh Tuan Luu"],"authors_zh":"Huimin Xu, Xin Mao, Feng-Lin Li, Xiaobao Wu, Wang Chen, Wei Zhang, Anh Tuan Luu","tracks":["training_usage_optimization_objectives","process_trace_supervision_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["preference_learning","process_supervision"],"construction_layer":["optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["process-supervision","preference-optimization","stepwise-reward","mathematical-reasoning","self-supervision"],"status":"verified","priority":"必读","paper_type_zh":"自监督逐步骤偏好优化研究","best_for_zh":"希望从推理 rollout 构造过程监督偏好信号、又不依赖 GPT-4 或人工步骤标签的读者。","confidence":"high","one_line":["Full-Step-DPO uses a self-supervised process reward model to score every reasoning step and weight preference optimization across the full chain.","Full-Step-DPO 使用自监督过程奖励模型为每个推理步骤打分，并在整条推理链上加权偏好优化。"],"why":"It changes both the construction and training consumer of reasoning preference data by retaining reward information from every step.","primary_link":"https://aclanthology.org/2025.findings-acl.1249/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Anna7355/Full-Step-DPO"}],"link_count":4,"sections":9},{"id":"game-rl-verifiable-game-data-2025","title":"Game-RL: Synthesizing Multimodal Verifiable Game Data to Boost VLMs' General Reasoning","year":2025,"venue":"ICLR 2026","authors":["Jingqi Tong","Jixin Tang","Hangcheng Li","Yurong Mou","Ming Zhang","Jun Zhao","Yanbo Wen","Fan Song","Jiahao Zhan","Yuyang Lu","Chaoran Tao","Zhiyuan Guo","Jizhou Yu","Tianhao Cheng","Zhiheng Xi","Changhao Jiang","Zhangyue Yin","Yining Zheng","Weifeng Ge","Guanhua Chen","Tao Gui","Xipeng Qiu","Qi Zhang","Xuanjing Huang"],"authors_zh":"Jingqi Tong, Jixin Tang, Hangcheng Li, Yurong Mou, Ming Zhang, Jun Zhao, Yanbo Wen, Fan Song, Jiahao Zhan, Yuyang Lu, Chaoran Tao, Zhiyuan Guo, Jizhou Yu, Tianhao Cheng, Zhiheng Xi, Changhao Jiang, Zhangyue Yin, Yining Zheng, Weifeng Ge, Guanhua Chen, Tao Gui, Xipeng Qiu, Qi Zhang, Xuanjing Huang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["vision-language","verifiable-data","game-reasoning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"可验证游戏任务的多模态推理数据合成与训练论文","best_for_zh":"需要可由游戏状态和代码复核的图像、步骤推理与答案记录来训练或评估视觉语言模型的研究者。","confidence":"high","one_line":["Game-RL synthesizes image-grounded game tasks whose answers can be checked by game code, yielding 140K multimodal reasoning records.","GameQA-140K 覆盖 30 个游戏、158 类可验证任务，每项带图像、步骤推理与标准答案，可从游戏代码复算反馈。"],"why":"It replaces uncheckable visual rationales with a game-state feedback contract.","primary_link":"https://arxiv.org/abs/2505.13886","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tongjingqi/Game-RL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenMOSS-Team/GameQA-140K"}],"link_count":5,"sections":9},{"id":"gateau-long-context-selection-2025","title":"GATEAU: Selecting Influential Samples for Long Context Alignment","year":2025,"venue":"EMNLP 2025","authors":["Shuzheng Si","Haozhe Zhao","Gang Chen","Yunshui Li","Kangyang Luo","Chuancheng Lv","Kaikai An","Fanchao Qi","Baobao Chang","Maosong Sun"],"authors_zh":"Shuzheng Si, Haozhe Zhao, Gang Chen, Yunshui Li, Kangyang Luo, Chuancheng Lv, Kaikai An, Fanchao Qi, Baobao Chang, Maosong Sun","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["long-context","instruction-tuning","reasoning"],"tags":["long-context","data-selection","sft","attention"],"status":"verified","priority":"必读","paper_type_zh":"长上下文指令数据选择研究","best_for_zh":"筛选必须教会真实长程依赖使用的长指令数据的读者。","confidence":"high","one_line":["GATEAU selects long SFT records using context-window perplexity gaps and attention-relevance alignment.","GATEAU 以上下文窗口困惑度差和注意力—相关性对齐选择长监督微调记录。"],"why":"It turns long-range dependence into an explicit criterion for which records are admitted to long-context SFT.","primary_link":"https://aclanthology.org/2025.emnlp-main.375/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/S1s-Z/GATEAU"}],"link_count":4,"sections":9},{"id":"gdpval-2025","title":"GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks","year":2025,"venue":"OpenAI research publication / arXiv preprint","authors":["Tejal Patwardhan","Rachel Dias","Elizabeth Proehl","Grace Kim","Michele Wang","Olivia Watkins","Simón Posada Fishman","Marwan Aljubeh","Phoebe Thacker","Laurance Fauconnet","Natalie S. Kim","Patrick Chao","Samuel Miserendino","Gildas Chabot","David Li","Michael Sharman","Alexandra Barr","Amelia Glaese","Jerry Tworek"],"authors_zh":"Tejal Patwardhan 等（OpenAI）","tracks":["benchmarks_evaluation_surfaces","judgment_rubric_domain_expert_data"],"source_role":["benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["professional-work-agents","economically-valuable-tasks","grading-service"],"tags":["benchmark","professional-work-evaluation","economically-valuable-tasks","grading-service"],"status":"verified","priority":"可读","paper_type_zh":"OpenAI 研究报告与 arXiv 的专业工作成果评测基准","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"medium","one_line":["GDPval evaluates economically valuable one-shot professional deliverables with expert blind comparisons and an experimental public grader.","GDPval 用专家盲评和实验性公开评分服务评测经济价值专业交付物。"],"why":"It evaluates economically valuable professional deliverables with expert comparison rather than short-answer scoring.","primary_link":"https://arxiv.org/abs/2510.04374","links":[{"key":"data","label":["Data","数据"],"url":"https://evals.openai.com/gdpval"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/openai/gdpval"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/gdpval/"}],"link_count":6,"sections":9},{"id":"gemini-2-5-computer-use-model-card-2025","title":"Gemini 2.5 Computer Use - Model Card","year":2025,"venue":"Google DeepMind model card","authors":["Google DeepMind"],"authors_zh":"Google DeepMind","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["agent_training","safety_alignment","evaluation","audit"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["agentic_systems","computer_use","web_navigation","mobile_control","multimodal","safety"],"tags":["gemini-2-5-computer-use","gemini","google-deepmind","computer-use","web-agent","screenshot-action","function-calling","human-judgment","safety-confirmation","prompt-injection","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿 computer-use agent 模型卡与环境披露台账","best_for_zh":"设计或审计截图型 web/mobile agent 数据、反馈契约、环境复现与安全确认机制的研究者","confidence":"high","one_line":["Gemini 2.5 Computer Use exposes a screenshot-action loop, human whole-trajectory judgments, partial environment details, and high-stakes safety gating, but not UI training data, reward, rollouts, or reproducible web pins.","Gemini 2.5 Computer Use Model Card 披露截图—动作—客户端执行闭环、整轨人类多数票和高风险动作安全门控，但未公开 UI 后训练数据、reward、rollout 或可复现的浏览器环境 pin。"],"why":"A public agent API and benchmark scores make deployment and evaluation concrete without revealing training trajectories, feedback objectives, or environment lineage.","primary_link":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Computer-Use-Model-Card.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/google-gemini/computer-use-preview"},{"key":"project","label":["Project","项目主页"],"url":"https://blog.google/innovation-and-ai/models-and-research/google-deepmind/gemini-computer-use-model/"}],"link_count":4,"sections":9},{"id":"gemini-2-5-deep-think-model-card-2025","title":"Gemini 2.5 Deep Think - Model Card","year":2025,"venue":"Google DeepMind model card","authors":["Google DeepMind"],"authors_zh":"Google DeepMind","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","safety_alignment","evaluation","audit","test_time_compute"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","theorem_proving","code","scientific_reasoning","multimodal","safety","cybersecurity"],"tags":["gemini-2-5-deep-think","gemini","google-deepmind","frontier-report","model-card","data-disclosure-ledger","parallel-thinking","reinforcement-learning","critic-feedback","test-time-compute","theorem-proving","safety-audit"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型卡与数据披露台账","best_for_zh":"审计前沿推理模型的数据对象、反馈契约、test-time compute、评测预算和版本映射的研究者","confidence":"high","one_line":["Deep Think discloses parallel hypothesis generation, added reasoning and mathematics data, novel RL, human/critic feedback, and frontier-safety tests, but not its records, reasoning reward, branch budget, or checkpoint lineage.","Gemini 2.5 Deep Think Model Card 披露并行假设生成、追加的推理与数学数据、novel RL、人类/critic 反馈和前沿安全评测，但未公开训练记录、推理奖励、分支预算或 checkpoint 谱系。"],"why":"It separates Deep Think-specific disclosures from family-level context and identifies unavailable data objects, feedback contracts, inference budgets, and safety artifacts.","primary_link":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Deep-Think-Model-Card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://blog.google/products-and-platforms/products/gemini/gemini-2-5-deep-think/"}],"link_count":3,"sections":9},{"id":"gemini-2-5-technical-report-2025","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","year":2025,"venue":"Google Technical Report","authors":["Gemini Team"],"authors_zh":"Gemini Team（Google）","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","preference_learning","reward_modeling","rlvr","agent_training","safety_alignment","test_time_compute","audit"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["reasoning","code","mathematics","multimodal","long_context","agentic_systems"],"tags":["gemini-2-5","gemini","frontier-report","data-disclosure-ledger","multimodal","thinking","sft","reward-modeling","rl","tool-use","safety-audit"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态推理技术报告与数据披露台账","best_for_zh":"需要审查前沿推理模型的数据对象、反馈契约、工具使用训练和审计边界的读者","confidence":"high","one_line":["Gemini 2.5 partially discloses multimodal pre/post-training categories, SFT/RM/RL, verifiable and model-generated rewards, tool-use RL, and safety audits, but withholds datasets, reward implementations, and reproducibility-critical construction detail.","Gemini 2.5 披露了多模态预训练/后训练类别、SFT/RM/RL、可验证与模型生成奖励、工具使用 RL 和安全审计，但未公开数据集、奖励实现及可复现所需的关键构造细节。"],"why":"It is a high-value Track 12 ledger item because it exposes real frontier post-training interfaces and the specific artifacts still unavailable for independent data, reward, and audit verification.","primary_link":"https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://blog.google/innovation-and-ai/models-and-research/google-deepmind/gemini-model-thinking-updates-march-2025/"}],"link_count":2,"sections":9},{"id":"gemini-robotics-1-5-2025","title":"Gemini Robotics 1.5: Pushing the Frontier of Generalist Robots with Advanced Embodied Reasoning, Thinking, and Motion Transfer","year":2025,"venue":"arXiv preprint","authors":["Gemini Robotics Team"],"authors_zh":"Gemini Robotics Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","agent_environment","benchmark","verifier_reward","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","evaluation","audit","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["robotics","embodied_reasoning","vision_language_action","robot_manipulation","multimodal_reasoning","spatial_reasoning","agentic_systems","safety"],"tags":["google-deepmind","gemini-robotics","motion-transfer","multi-embodiment","embodied-reasoning","vision-language-action","robot-sensor-action-data","natural-language-thinking","success-detection","progress-estimation","auto-red-teaming","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"多具身 VLA 技术报告与训练数据披露台账","best_for_zh":"研究多机器人迁移、embodied reasoning、自然语言 thinking、物理智能体评测与安全数据的读者","confidence":"high","one_line":["Gemini Robotics 1.5 discloses ALOHA, Franka, and Apollo sensor/action data mixed with Internet text-image-video data, cross-embodiment Motion Transfer, natural-language thinking before action, and progress/success contracts, but not the mixture, record schema, labels, optimizer, rights, or released corpus.","Gemini Robotics 1.5 披露 ALOHA、Franka、Apollo 机器人数据与互联网多模态数据的 Motion Transfer、动作前自然语言 thinking 以及 progress/success/failure 评测契约，但未开放算法细节、thinking 标签来源、数据权利或训练工件。"],"why":"It is a high-value disclosure ledger for physical agents because it connects frontier-model fine-tuning to multi-robot trajectories, synthetic captioning, tool-using orchestration, inference-time thinking, environment feedback, and adversarial safety data while making the reproducibility and deployment gaps explicit.","primary_link":"https://arxiv.org/abs/2510.03342","links":[{"key":"project","label":["Project","项目主页"],"url":"https://deepmind.google/blog/gemini-robotics-15-brings-ai-agents-into-the-physical-world/"}],"link_count":3,"sections":9},{"id":"gemini-robotics-physical-world-2025","title":"Gemini Robotics: Bringing AI into the Physical World","year":2025,"venue":"arXiv preprint","authors":["Gemini Robotics Team"],"authors_zh":"Gemini Robotics Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","benchmark","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["robotics","embodied_reasoning","vision_language_action","robot_manipulation","multimodal_reasoning","spatial_reasoning"],"tags":["google-deepmind","gemini-robotics","embodied-reasoning","vision-language-action","robot-demonstrations","teleoperation","action-chunks","trajectory-relabeling","specialization","erqa","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿机器人 VLA 技术报告与数据披露台账","best_for_zh":"研究机器人示范数据、embodied reasoning、VLA specialization 与真实环境评测契约的读者","confidence":"high","one_line":["Gemini Robotics discloses thousands of hours of ALOHA 2 expert demonstrations, image-and-instruction inputs, action-chunk outputs, trajectory-relabelled reasoning data, and 2,000–5,000-episode specialization sets, but not the robot corpus, mixture, optimizer, or Gemini training-step count.","Gemini Robotics 披露了 ALOHA 2 专家示范、图像加自然语言指令到 action chunks 的 VLA 对象、未来约 1 秒双臂轨迹中间量以及每任务 2,000–5,000 条 specialization demos，但未开放专有机器人语料、训练步数或 optimizer。"],"why":"It provides an unusually concrete disclosure ledger for moving from internet-scale multimodal reasoning to physical action, while showing that released ERQA evaluation records and proprietary action-training episodes have very different reuse and audit status.","primary_link":"https://arxiv.org/abs/2503.20020","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/embodiedreasoning/ERQA"},{"key":"project","label":["Project","项目主页"],"url":"https://deepmind.google/blog/gemini-robotics-brings-ai-into-the-physical-world/"}],"link_count":4,"sections":9},{"id":"gemma-3-technical-report-2025","title":"Gemma 3 Technical Report","year":2025,"venue":"arXiv preprint","authors":["Gemma Team"],"authors_zh":"Gemma Team（Google DeepMind）","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","distillation","preference_learning","reward_modeling","rlvr","evaluation","audit","safety_alignment"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["language_modeling","multimodal_reasoning","mathematical_reasoning","code_generation","instruction_following","multilingual","safety_alignment","privacy"],"tags":["google-deepmind","gemma-3","open-weights","multimodal","knowledge-distillation","sparse-logits","instruction-tuning","human-feedback","reward-model","code-execution-reward","math-ground-truth","bond","warm","warp","memorization-audit","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"开放权重前沿模型技术报告与数据反馈台账","best_for_zh":"研究知识蒸馏、human-feedback RM、程序化奖励、QAT、记忆审计与开放权重许可边界的读者","confidence":"high","one_line":["Gemma 3 discloses 2T-14T pretraining, 256 teacher logits per token, large-IT-teacher distillation, mixed human/code/math RL rewards, QAT, and open weights, but not corpora, preferences, RMs, teachers, rollouts, or verifiers.","Gemma 3 披露 1B/4B/12B/27B 的 2T–14T token 预训练、每 token 256 个 teacher logits、large-IT-teacher 蒸馏、human/code/math 混合 reward、QAT 与开放权重，但未公开 corpus、preference、RM、teacher、rollout 或 verifier。"],"why":"It shows that open weight access is much broader than the underlying data, distillation, reward, and verifier disclosure.","primary_link":"https://arxiv.org/abs/2503.19786","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/google/gemma_pytorch"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/google/gemma-3-27b-it"},{"key":"project","label":["Project","项目主页"],"url":"https://ai.google.dev/gemma/docs/core/model_card_3"}],"link_count":5,"sections":9},{"id":"general-reasoner-2025","title":"General-Reasoner: Advancing LLM Reasoning Across All Domains","year":2025,"venue":"arXiv preprint (2025)","authors":["Xueguang Ma","Qian Liu","Dongfu Jiang","Ge Zhang","Zejun MA","Wenhu Chen"],"authors_zh":"Xueguang Ma、Qian Liu、Dongfu Jiang、Ge Zhang、Zejun MA、Wenhu Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English physics, chemistry, finance, electronics, mathematics, and broad web knowledge"],"tags":["instruction-demonstration-rationale","arxiv-2505.14652","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"跨领域推理强化学习","confidence":"high","one_line":["General-Reasoner crawls broad questions, normalizes answer types, and uses a context-aware generative verifier to retain questions with scoreable answers.","General-Reasoner 从网页整理 22.8 万条跨领域可验证问答，并用生成式验证器处理多种等价答案。"],"why":"Rule-based answer checkers work for math but reject equivalent free-form answers in science and professional domains, blocking broad RL data collection.","primary_link":"https://arxiv.org/abs/2505.14652","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TIGER-Lab/WebInstruct-verified"}],"link_count":2,"sections":9},{"id":"probabilistic-uncertainty-comparative-judge-2025","title":"Generalised Probabilistic Modelling and Improved Uncertainty Estimation in Comparative LLM-as-a-judge","year":2025,"venue":"UAI 2025 (Poster)","authors":["Yassir Fathullah","Mark J. F. Gales"],"authors_zh":"Yassir Fathullah、Mark J. F. Gales","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"比较式 LLM judge 的概率建模、主动比较与不确定性评测研究","best_for_zh":"需要以较少成对比较获得稳定候选排序并量化置信度的研究者。","confidence":"medium","one_line":["Develops uncertainty estimates for comparative LLM judging and ranking stability.","用广义概率模型与重排序概率，减少成对 LLM judge 排名所需比较。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=YJ4gr1RT37","links":[],"link_count":1,"sections":9},{"id":"leannavigator-state-graph-theorems-2025","title":"Generating Millions Of Lean Theorems With Proofs By Exploring State Transition Graphs","year":2025,"venue":"arXiv","authors":["David Yin","Jing Gao"],"authors_zh":"David Yin, Jing Gao","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","theorem-proving"],"tags":["lean4","formal-proofs","state-graphs","synthetic-data","2025"],"status":"verified","priority":"可读","paper_type_zh":"Lean 形式化证明数据集与状态图生成方法","best_for_zh":"需要大规模 Lean 训练语料、形式化验证器或定理证明评测数据的研究者。","confidence":"high","one_line":["LeanNavigator explores Mathlib4 state graphs to release 4.7 million Lean theorem–proof pairs whose proofs are accepted by Lean.","LeanNavigator 通过遍历 Mathlib4 的状态转移图生成 470 万条 Lean 定理—证明对，并以 Lean 检查器保证证明可验证。"],"why":"The release makes formal proof acceptance a reproducible terminal predicate for a large synthetic theorem corpus.","primary_link":"https://arxiv.org/abs/2503.04772","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/davidsyin/leannavigator"},{"key":"data","label":["Data","数据"],"url":"https://zenodo.org/records/13989482"}],"link_count":4,"sections":9},{"id":"generative-verifiers-next-token-prediction-2025","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","year":2025,"venue":"ICLR","authors":["Lunjun Zhang","Arian Hosseini","Hritik Bansal","Seyed Mehran Kazemi","Aviral Kumar","Rishabh Agarwal"],"authors_zh":"Lunjun Zhang；Arian Hosseini；Hritik Bansal；Seyed Mehran Kazemi；Aviral Kumar；Rishabh Agarwal","tracks":["data_construction_open_release_recipes"],"source_role":["verifier_reward","construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","algorithmic-reasoning"],"tags":["generative-verifier","reward-modeling","verification-rationales","synthetic-critiques","reference-guided-grading","majority-voting"],"status":"partial","priority":"必读","paper_type_zh":"生成式 verifier 与合成验证理由数据构建研究","best_for_zh":"需要设计 reward model、生成验证理由、Best-of-N 选择器或审计开放反馈数据的研究者","confidence":"high","one_line":["Generative Verifiers turns answer verification into next-token SFT over direct verdicts or generated critiques and releases GSM8K feedback records, but full license scope, code, checkpoints, and decontamination remain unresolved.","Generative Verifiers 将答案验证改写为直接判决或生成式 critique 的下一 token SFT，并公开 GSM8K 反馈记录，但生成理由的许可证范围、代码、checkpoint 与去污染仍未解决。"],"why":"It makes the verifier's language target, teacher, reference guidance, filtering rule, and inference budget explicit data-design choices that can be audited independently of Best-of-N scores.","primary_link":"https://arxiv.org/abs/2408.15240","links":[{"key":"data","label":["Data","数据"],"url":"https://github.com/genrm-star/genrm-critiques"},{"key":"project","label":["Project","项目主页"],"url":"https://sites.google.com/view/generative-reward-models"}],"link_count":5,"sections":9},{"id":"genius-unsupervised-self-training-2025","title":"Genius: A Generalizable and Purely Unsupervised Self-Training Framework For Advanced Reasoning","year":2025,"venue":"ACL","authors":["Fangzhi Xu","Hang Yan","Chang Ma","Haiteng Zhao","Qiushi Sun","Kanzhi Cheng","Junxian He","Jun Liu","Zhiyong Wu"],"authors_zh":"Fangzhi Xu、Hang Yan、Chang Ma、Haiteng Zhao、Qiushi Sun、Kanzhi Cheng、Junxian He、Jun Liu、Zhiyong Wu","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","process_supervision","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","pairwise_preference","trajectory_value"],"training_use":["preference_learning","process_supervision"],"construction_layer":["prompt_sourcing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["general-reasoning","mathematics","code"],"tags":["unsupervised-self-training","self-reward","foresight-sampling","step-level-preferences","aco","model-release"],"status":"partial","priority":"可读","paper_type_zh":"无监督自我训练与步骤级偏好构造方法","best_for_zh":"研究推理数据构造、自奖励、步骤级偏好学习与数据审计的读者","confidence":"medium","one_line":["Genius converts general queries into same-policy foresight-ranked trajectory pairs for ACO, but the score is not a correctness verifier and the exact preference data is not released.","Genius 将通用查询转换为由同一策略的未来 continuation 分数排序的步骤级轨迹偏好并用于 ACO；该分数不是正确性 verifier，完整偏好数据也未发布。"],"why":"It exposes a label-free construction path from queries to step-local preferences and a policy checkpoint, while showing why lineage, proxy validation, code-version pinning, and rejected-sample retention are necessary for reasoning-data reuse.","primary_link":"https://aclanthology.org/2025.acl-long.644/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xufangzhi/Genius"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/xufangzhi/genius"}],"link_count":6,"sections":9},{"id":"genprm-scaling-test-time-compute-process-reward-models-2025","title":"GenPRM: Scaling Test-Time Compute of Process Reward Models via Generative Reasoning","year":2025,"venue":"AAAI 2026","authors":["Jian Zhao","Runze Liu","Kaiyan Zhang","Zhimu Zhou","Junqi Gao","Dong Li","Jiafei Lyu","Zhouyi Qian","Biqing Qi","Xiu Li","Bowen Zhou"],"authors_zh":"Jian Zhao、Runze Liu、Kaiyan Zhang、Zhimu Zhou、Junqi Gao 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["math_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"生成式过程奖励模型与数学过程监督数据论文","best_for_zh":"构建可解释的数学步骤判定器，或研究 PRM 测试时扩展。","confidence":"high","one_line":["GenPRM releases reasoning-and-verification process supervision that lets generative reward models scale judgments at test time.","GenPRM 发布带推理与验证的过程监督数据，使生成式奖励模型能够在测试时扩展判定。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://ojs.aaai.org/index.php/AAAI/article/view/40797","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RyanLiu112/GenPRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GenPRM/GenPRM-MATH-Data"},{"key":"project","label":["Project","项目主页"],"url":"https://ryanliu112.github.io/GenPRM"}],"link_count":5,"sections":9},{"id":"genselect-best-of-n-2025","title":"GenSelect: A Generative Approach to Best-of-N","year":2025,"venue":"2nd AI for Math Workshop @ ICML 2025 (poster)","authors":["Shubham Toshniwal","Ivan Sorokin","Aleksander Ficek","Ivan Moshkov","Igor Gitman"],"authors_zh":"Shubham Toshniwal、Ivan Sorokin、Aleksander Ficek、Ivan Moshkov、Igor Gitman","tracks":["rollout_search_test_time_trace_data"],"source_role":["verifier_reward","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","scaling_report"],"domains":["mathematics","competition_mathematics"],"tags":["genselect","best-of-n","generative-verifier","n-ary-tournament","self-selection","selector-trace","rejected-candidates","test-time-compute","competition-mathematics"],"status":"partial","priority":"可读","paper_type_zh":"生成式 Best-of-N 选择与测试时扩展研究","best_for_zh":"研究 rollout 选择器、生成式 verifier、测试时算力核算与候选轨迹审计的读者","confidence":"high","one_line":["GenSelect asks a reasoning model to compare indexed solution summaries, emit a chosen candidate index, and scale selection over 64-solution pools through repeated N-ary knockout tournaments; a related official release exposes 565,620 packed selector rows.","GenSelect 让推理模型联合比较带索引的数学解答摘要并生成所选索引，再用 N 叉淘汰赛扩展到 64 个候选；其论文实验轨迹未发布，565,620 条公开记录来自相关但不同的 OpenMathReasoning split。"],"why":"It separates candidate generation, comparative selector reasoning, chosen index, tournament budget, and final correctness, while the difference between the unreleased paper evaluation pools and the packed public training split shows exactly which rejected candidates, labels, and compute records remain auditable.","primary_link":"https://openreview.net/forum?id=8LhnmNmUDb","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Skills/tree/main/recipes/openmathreasoning/scripts/genselect"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenMathReasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://nvidia-nemo.github.io/Skills/releases/openreasoning/evaluation/"}],"link_count":7,"sections":9},{"id":"glm-4-5-arc-foundation-models-2025","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","year":2025,"venue":"arXiv preprint","authors":["GLM-4.5 Team"],"authors_zh":"GLM-4.5 Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward","agent_environment","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","preference_learning","reward_modeling","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["general_reasoning","mathematics","coding","agent","web_search","tool_use"],"tags":["glm-4-5","zai","frontier-report","data-disclosure-ledger","reasoning","coding","agent-training","reinforcement-learning","reward-model","tool-use"],"status":"partial","priority":"必读","paper_type_zh":"前沿 ARC 基座模型技术报告与数据披露账本","best_for_zh":"需要审计前沿推理、代码和 agent 模型的数据来源、反馈契约、轨迹筛选、环境依赖与发布边界的读者","confidence":"medium","one_line":["GLM-4.5 discloses a broad 23T-token ARC training and post-training pipeline with expert self-distillation, verifier/reward/judge feedback, and agent environments, but releases weights and inference materials rather than the source data, RM/Judge artifacts, replayable environments, or audit records needed for independent verification.","GLM-4.5 披露了约 23T token 的 ARC 训练与后训练框架，包含专家自蒸馏、验证器/奖励/裁判反馈和 agent 环境；但官方发布的是权重与推理材料，而非原始语料、RM/Judge 工件、可回放环境或审计记录。"],"why":"It makes the distinction Track 12 needs: a detailed frontier-model narrative and MIT model release can reveal useful high-level data and feedback interfaces, yet they do not establish access to the original data, reward/judge contract, environment state, provenance, licenses, decontamination evidence, or reproducible training process.","primary_link":"https://arxiv.org/abs/2508.06471","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zai-org/GLM-4.5"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zai-org/GLM-4.5"},{"key":"project","label":["Project","项目主页"],"url":"https://docs.z.ai/guides/llm/glm-4.5"}],"link_count":6,"sections":9},{"id":"glm-4-1v-4-5v-2025","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["GLM-V Team"],"authors_zh":"GLM-V Team 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["vision_language","gui_agents","video","long_documents","stem"],"tags":["frontier-report","multimodal-reasoning","rlcs","rlhf","rlvr","long-cot","gui-agents","reward-system","open-weights","disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"多模态推理模型技术报告","best_for_zh":"研究多模态 RL 与披露边界的读者","confidence":"medium","one_line":["GLM-V discloses a broad multimodal pre-training, long-CoT SFT and GRPO/RLCS pipeline with domain-specific rewards, and releases weights plus reward/inference code while withholding the training records and curriculum ledger.","GLM-V 报告以 RLCS 扩展多模态推理，但样本与奖励来源仍未充分披露。"],"why":"It is a useful disclosure-ledger case because model and verifier availability can be audited separately from unreleased multimodal sources, long-CoT traces, difficulty labels, rollouts and selection decisions.","primary_link":"https://arxiv.org/abs/2507.01006","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zai-org/GLM-V"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zai-org/GLM-4.5V"}],"link_count":5,"sections":9},{"id":"gm-prm-generative-multimodal-process-reward-model-2025","title":"GM-PRM: A Generative Multimodal Process Reward Model for Multimodal Mathematical Reasoning","year":2025,"venue":"Proceedings of the 4th Workshop on Advances in Language and Vision Research (ALVR 2026)","authors":["Jianghangfan Zhang","Yibo Yan","Kening Zheng","Xin Zou","Song Dai","Xuming Hu"],"authors_zh":"Jianghangfan Zhang、Yibo Yan、Kening Zheng、Xin Zou、Song Dai、Xuming Hu","tracks":["process_trace_supervision_data","training_usage_optimization_objectives"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal","math_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"生成式多模态过程奖励模型与纠错监督数据论文","best_for_zh":"研究视觉数学推理的步骤诊断、纠错和测试时重排序。","confidence":"high","one_line":["GM-PRM releases 20K multimodal reasoning records that teach a process model to diagnose and correct the first erroneous visual-math step.","GM‑PRM 发布 2 万条多模态推理记录，训练过程模型诊断并修正首个视觉数学错误步骤。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://aclanthology.org/2026.alvr-main.11/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zijinghuafen/GM-PRM-20K"}],"link_count":3,"sections":9},{"id":"goedel-prover-v2-2025","title":"Goedel-Prover-V2: Scaling Formal Theorem Proving with Scaffolded Data Synthesis and Self-Correction","year":2025,"venue":"arXiv preprint","authors":["Yong Lin","Shange Tang","Bohan Lyu","Ziran Yang","Jui-Hui Chung","Haoyu Zhao","Lai Jiang","Yihan Geng","Jiawei Ge","Jingruo Sun","Jiayun Wu","Jiri Gesi","Ximing Lu","David Acuna","Kaiyu Yang","Hongzhou Lin","Yejin Choi","Danqi Chen","Sanjeev Arora","Chi Jin"],"authors_zh":"Yong Lin 等","tracks":["frontier_reports_data_disclosure_ledger","programmatically_verifiable_outcome_data"],"source_role":["model_report","construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["mathematics","formal_theorem_proving","lean"],"tags":["goedel-prover-v2","lean","formal-reasoning","self-correction","synthetic-data","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿形式化定理证明技术报告与数据披露账本","best_for_zh":"需要审计合成形式化数据、Lean 反馈、自我修正与后训练披露边界的读者","confidence":"high","one_line":["Goedel-Prover-V2 builds Lean proof models through verified whole-proof SFT, compiler-error self-correction data, scaffolded theorem synthesis, hybrid GRPO, and model averaging; public models and evaluation artifacts exist, but the core S1/S2/S3 training records are not released.","Goedel-Prover-V2 报告了支架式合成任务生成、基于 Lean 编译器反馈的自我修正、expert iteration 与强化学习；但本 Card 尚未核验其逐条数据、验证环境及具体发布工件。"],"why":"For the frontier disclosure track, it exposes how theorem statements, long proof traces, compiler feedback, and verifier rewards enter distinct post-training stages while preserving the gap between formal acceptance, semantic faithfulness, and release completeness.","primary_link":"https://arxiv.org/abs/2508.03613","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Goedel-LM/Goedel-Prover-V2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Goedel-LM/MathOlympiadBench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Goedel-LM/Goedel-Prover-V2-32B"},{"key":"project","label":["Project","项目主页"],"url":"https://blog.goedel-prover.com/"}],"link_count":7,"sections":9},{"id":"goedel-prover-2025","title":"Goedel-Prover: A Frontier Model for Open-Source Automated Theorem Proving","year":2025,"venue":"arXiv technical report","authors":["Yong Lin","Shange Tang","Bohan Lyu","Jiayun Wu","Hongzhou Lin","Kaiyu Yang","Jia Li","Mengzhou Xia","Danqi Chen","Sanjeev Arora","Chi Jin"],"authors_zh":"Yong Lin、Shange Tang、Bohan Lyu、Jiayun Wu、Hongzhou Lin、Kaiyu Yang、Jia Li、Mengzhou Xia、Danqi Chen、Sanjeev Arora、Chi Jin","tracks":["instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","preference_learning","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["formal-mathematics","lean4","automated-theorem-proving"],"tags":["instruction-demonstration-rationale","formal-math","lean4","expert-iteration","arxiv-2502.07640","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放形式数学数据集、构造流程与证明器报告","best_for_zh":"适合构建 Lean 完整证明语料、自动形式化流程和验证器驱动的专家迭代。","confidence":"high","one_line":["Goedel-Prover releases large-scale informal/formal Lean statements and 29,750 complete proofs produced through compiler-verified expert iteration.","Goedel-Prover 公开大规模自然语言与 Lean 陈述，以及通过编译器核验的 29,750 条完整证明。"],"why":"It exposes statement style, faithfulness checks, proof search, compiler verification, and cumulative SFT as distinct data-quality decisions.","primary_link":"https://arxiv.org/abs/2502.07640","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Goedel-LM/Goedel-Prover"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Goedel-LM/Goedel-Pset-v1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/Goedel-LM/Lean-workbook-proofs"},{"key":"project","label":["Project","项目主页"],"url":"https://goedel-lm.github.io/"}],"link_count":7,"sections":9},{"id":"gpt-5-system-card-2025","title":"GPT-5 System Card","year":2025,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["evaluation","safety_alignment"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety"],"tags":["openai","gpt-5","system-card","router","safe-completions","reinforcement-learning","test-time-compute","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿系统卡与数据披露账本","best_for_zh":"需要审计路由训练、安全完成、测试时计算及前沿模型披露边界的读者","confidence":"high","one_line":["The GPT-5 System Card discloses broad source classes, reinforcement-learning reasoning, continuous router training on model switches, preference rates, and measured correctness, safe-completions, and parallel test-time compute, but not the underlying records or reproducible reward and data contracts.","GPT-5 System Card 报告路由器持续利用模型切换、偏好率和测得正确性训练，以及 safe-completions 和 thinking-pro 的并行测试时计算；其支撑数据和训练配方仍未披露。"],"why":"It offers a useful Track 12 ledger of a routed frontier system while making visible the boundary between post-training disclosure, safety evaluation, deployment safeguards, and benchmark evidence.","primary_link":"https://openai.com/index/gpt-5-system-card/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://deploymentsafety.openai.com/gpt-5"}],"link_count":2,"sections":9},{"id":"openai-gpt-5-1-codex-max-system-card-2025","title":"GPT-5.1-Codex-Max System Card","year":2025,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["agent_training","evaluation","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["software_engineering","coding_agents","cybersecurity","safety","biosecurity","ai_self_improvement"],"tags":["openai","gpt-5-1-codex-max","system-card","coding-agent","compaction","long-horizon-agent","safety-training","prompt-injection","destructive-action","reinforcement-learning","environment-reward","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿编码代理系统卡与数据披露账本","best_for_zh":"审计闭源编码代理训练对象、反馈契约、环境评测和披露边界的研究者","confidence":"high","one_line":["The GPT-5.1-Codex-Max System Card exposes synthetic malware and prompt-injection safety data plus RL rollouts where a user model makes conflicting edits and preservation earns positive reinforcement, but not the record corpus, reward implementation, source mixture, or train/evaluation lineage.","GPT-5.1-Codex-Max 系统卡披露了合成恶意软件与 prompt-injection 安全数据，以及由 user model 注入冲突编辑、以保留用户改动获得正向强化的 RL episode；其主要价值是可审计披露边界，而非提供可复用训练语料或 reward 实现。"],"why":"It is an unusually concrete frontier-report ledger for coding-agent post-training because it links prompts, environment changes, episode behavior, rewards, compaction budgets, and executable evaluation predicates while clearly demonstrating how much remains unavailable for independent reuse or attribution.","primary_link":"https://openai.com/index/gpt-5-1-codex-max-system-card/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://deploymentsafety.openai.com/gpt-5-1-codex-max"}],"link_count":2,"sections":9},{"id":"gpt-oss-model-card-2025","title":"gpt-oss-120b & gpt-oss-20b Model Card","year":2025,"venue":"arXiv preprint","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["sft","distillation","rlvr","agent_training","safety_alignment"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","coding","science","agentic_tool_use","safety"],"tags":["gpt-oss","frontier-report","data-disclosure-ledger","open-weights","model-card","harmony","chain-of-thought","tool-use","safety-alignment"],"status":"partial","priority":"必读","paper_type_zh":"前沿开源权重模型卡与数据披露台账","best_for_zh":"需要辨析开放模型发布与未公开后训练数据、反馈机制及审计证据边界的读者","confidence":"high","one_line":["gpt-oss discloses broad pretraining, CoT-RL/tool interfaces, safety mitigation, and Apache-2.0 weights/tool references while withholding source-level training, reward, rollout, and audit artifacts.","gpt-oss 披露了概括性的预训练、CoT-RL/工具接口、安全缓解措施以及 Apache-2.0 权重与工具参考实现，但未公开按来源划分的训练数据、奖励、rollout 或审计工件。"],"why":"It distinguishes a real open-weight inference release from an unreleased post-training pipeline whose data provenance, reasoning traces, reward contract, and contamination evidence cannot be audited.","primary_link":"https://arxiv.org/abs/2508.10925","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/gpt-oss"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/openai/gpt-oss-120b"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-gpt-oss/"}],"link_count":6,"sections":9},{"id":"openai-gpt-oss-safeguard-technical-report-2025","title":"gpt-oss-safeguard technical report","year":2025,"venue":"OpenAI technical report","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["safety_alignment","evaluation"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["safety","content_moderation","reasoning"],"tags":["openai","gpt-oss-safeguard","technical-report","frontier-report","data-disclosure-ledger","safety-classification","policy-reasoning","chain-of-thought","open-weights"],"status":"partial","priority":"可读","paper_type_zh":"开放权重安全推理模型技术报告与数据披露台账","best_for_zh":"需要区分开放权重、推理时策略分类接口与未公开后训练数据/反馈管线的安全、对齐和数据审计读者","confidence":"high","one_line":["OpenAI's gpt-oss-safeguard report describes open-weight models that classify developer-supplied policy/content pairs with chain-of-thought and a high-level human-judgment policy-labelling lineage, while withholding training records, reward details, and reproducibility artifacts.","OpenAI 的 gpt-oss-safeguard 技术报告披露了按开发者提供的策略和内容进行分类的开放权重模型、推理链和高层次的人类判断策略标注脉络，但未公开训练记录、奖励细节或可复现审计材料。"],"why":"It is a concrete Track 12 case where released weights and transparent inference interfaces coexist with material gaps in data provenance, policy-label supervision, reward calibration, and evaluation auditability.","primary_link":"https://cdn.openai.com/pdf/08b7dee4-8bc6-4955-a219-7793fb69090c/Technical_report__Research_Preview_of_gpt_oss_safeguard.pdf","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/openai/gpt-oss-safeguard"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-gpt-oss-safeguard/"}],"link_count":3,"sections":9},{"id":"grip-2025","title":"GRIP: A Graph-Based Reasoning Instruction Producer","year":2025,"venue":"NeurIPS 2025","authors":["Jiankang Wang","Jianjun Xu","Xiaorui Wang","Yuxin Wang","Mengting Xing","Shancheng Fang","Hongtao Xie"],"authors_zh":"Jiankang Wang, Jianjun Xu, Xiaorui Wang, Yuxin Wang, Mengting Xing, Shancheng Fang, Hongtao Xie","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematical_reasoning"],"tags":["grip","grip-math","synthetic-data","concept-graph","mathematical-reasoning","multi-model-filtering","instruction-tuning","partial-release"],"status":"partial","priority":"可读","paper_type_zh":"图引导数学推理数据构造与部分发布","best_for_zh":"研究合成推理数据、概念图重组、多模型筛选及发布审计的读者","confidence":"medium","one_line":["GRIP reports expanding 7.5K MATH seeds into 2,123,345 graph-conditioned question-solution pairs and officially releases 24,957 four-field rows, while code, the full corpus, judge weights, lineage, and licenses remain unavailable.","GRIP 报告把 7.5K 条 MATH 种子扩展为 2,123,345 个图条件问题—解答对，并通过官方补充材料发布 24,957 条四字段样本；代码、完整语料、评审权重、谱系和许可证仍未公开。"],"why":"The work makes graph relationship type, difficulty-routed generation, and separate question/solution gates explicit construction variables; its partial release also exposes the difference between a documented recipe, an inspectable sample, and an audit-ready corpus.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/6f1989abe9562c5cd306e070725fe0a3-Abstract-Conference.html","links":[{"key":"data","label":["Data","数据"],"url":"https://proceedings.neurips.cc/paper_files/paper/2025/file/6f1989abe9562c5cd306e070725fe0a3-Supplemental-Conference.zip"}],"link_count":5,"sections":9},{"id":"xai-grok-4-fast-model-card-2025","title":"Grok 4 Fast Model Card","year":2025,"venue":"xAI model card","authors":["xAI"],"authors_zh":"xAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","agentic_tool_use"],"tags":["xai","grok-4-fast","model-card","frontier-report","data-disclosure-ledger","safety-training","rlvr","tool-use","test-time-compute"],"status":"partial","priority":"可读","paper_type_zh":"闭源前沿模型卡与数据披露台账","best_for_zh":"审计和比较闭源前沿模型的后训练、工具使用、安全与披露边界的读者","confidence":"high","one_line":["Grok 4 Fast's model card and launch announcement disclose broad pretraining sources, de-duplication/classification, SFT and RL with human feedback, verifiable rewards and model grading, safety mitigations, and tool-use RL, but not record-level data, reward, filter, or evaluator contracts.","Grok 4 Fast 的模型卡和发布公告披露了宽泛的预训练来源、去重/分类、SFT、使用人类反馈/可验证奖励/模型评分的 RL、安全缓解措施及工具使用 RL，但未披露记录级数据、奖励、过滤器或评估器合约。"],"why":"It provides a closed-frontier disclosure baseline for separating report-level evidence about post-training, safety, tool use, and test-time behavior from missing provenance, data-rights, verifier, evaluation, and reproducibility evidence.","primary_link":"https://data.x.ai/2025-09-19-grok-4-fast-model-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://x.ai/news/grok-4-fast"}],"link_count":2,"sections":9},{"id":"xai-grok-4-model-card-2025","title":"Grok 4 Model Card","year":2025,"venue":"xAI model card","authors":["xAI"],"authors_zh":"xAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","agentic_tool_use"],"tags":["xai","grok-4","model-card","frontier-report","data-disclosure-ledger","safety-training","rlvr","tool-use"],"status":"partial","priority":"可读","paper_type_zh":"闭源前沿模型卡与数据披露台账","best_for_zh":"审计和比较闭源前沿模型的后训练、工具使用与安全披露边界的读者","confidence":"high","one_line":["Grok 4's model card and release announcement disclose broad pretraining sources, de-duplication/classification, SFT/RL with human feedback, verifiable rewards and model grading, and tool-use RL, but not record-level data, reward, filter, or evaluator contracts.","Grok 4 的模型卡与发布公告披露了宽泛的预训练来源、去重/分类、使用人类反馈、可验证奖励和模型评分的 SFT/RL，以及工具使用 RL，但未披露记录级数据、奖励、过滤器或评估器合约。"],"why":"It makes a closed frontier model's post-training and safety pipeline partly legible while preserving the distinction between high-level disclosure and missing provenance, data rights, verifier/reward specifics, evaluation controls, and reproducibility evidence.","primary_link":"https://data.x.ai/2025-08-20-grok-4-model-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://x.ai/news/grok-4"}],"link_count":2,"sections":9},{"id":"xai-grok-4-1-model-card-2025","title":"Grok 4.1 Model Card","year":2025,"venue":"xAI model card","authors":["xAI"],"authors_zh":"xAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","safety_alignment","evaluation"],"construction_layer":["reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","agentic_tool_use"],"tags":["xai","grok-4-1","model-card","frontier-report","data-disclosure-ledger","safety-training","model-based-reward","production-evaluation"],"status":"partial","priority":"可读","paper_type_zh":"闭源前沿模型卡与数据披露台账","best_for_zh":"需要区分模型安全评估、生产评估与可复用后训练数据证据的读者","confidence":"high","one_line":["Grok 4.1's official card names broad data categories and SFT/RL feedback families plus safety-filter training, while its announcement adds model-based rewards and production-rollout evaluation; neither source releases reusable training data or reward contracts.","Grok 4.1 的官方模型卡披露了宽泛数据类别、定向中期训练以及基于人类反馈、可验证奖励和模型评分器的 SFT/RL；发布公告补充模型奖励与生产流量评估，但二者均不构成可复用训练数据或奖励合约的证据。"],"why":"It distinguishes a useful official disclosure ledger from evidence of reusable post-training data: the sources support high-level pipeline and evaluation claims but leave provenance, rights, verifier details, production governance, and reproducibility unknown.","primary_link":"https://data.x.ai/2025-11-17-grok-4-1-model-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://x.ai/news/grok-4-1"}],"link_count":2,"sections":9},{"id":"groundedprm-2025","title":"GroundedPRM: Tree-Guided and Fidelity-Aware Process Reward Modeling for Step-Level Reasoning","year":2025,"venue":"arXiv","authors":["Yao Zhang","Yu Wu","Haowei Zhang","Weiguo Li","Haokun Chen","Jingpei Wu","Guohao Li","Zhen Han","Volker Tresp"],"authors_zh":"Yao Zhang、Yu Wu、Haowei Zhang、Weiguo Li、Haokun Chen、Jingpei Wu、Guohao Li、Zhen Han、Volker Tresp","tracks":["rollout_search_test_time_trace_data","training_usage_optimization_objectives","preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["process_supervision","verifier_reward","construction_recipe"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","test_time_compute"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer"],"domains":["math","reasoning"],"tags":["primary-link-checked","process-reward","mcts","tree-search","tool-validation","test-time-search","release-unverified"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹、搜索或测试时计算研究","best_for_zh":"需要审计 rollout、选择器、预算、数据谱系与复现边界的读者","confidence":"medium","one_line":["GroundedPRM converts MATH MCTS trajectories into about 40K generative step-supervision records by combining Wolfram Alpha checks with final-answer outcomes, but releases no author-confirmed tree, data, code, model, or run manifest.","该条目将推理轨迹、搜索选择或测试时计算作为可审计的研究对象；未披露字段已明确标为 unknown。"],"why":"It is a concrete example of turning local tool checks and global search outcomes into process-reward data, while exposing how verifier coverage, hybrid-credit parameters, filtered failures, and candidate budgets condition the resulting labels.","primary_link":"https://arxiv.org/abs/2510.14942","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/YaoZ720/GroundedPRMCode"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Yuuuuuu98/Grounded_PRM"},{"key":"project","label":["Project","项目主页"],"url":"https://yaozhang.ai/groundedprm/"}],"link_count":7,"sections":9},{"id":"grouse-evaluator-benchmark-2025","title":"GroUSE: A Benchmark to Evaluate Evaluators in Grounded Question Answering","year":2025,"venue":"COLING 2025","authors":["Sacha Muller","António Loison","Bilel Omrani","Gautier Viaud"],"authors_zh":"Sacha Muller, António Loison, Bilel Omrani, Gautier Viaud","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["A 144-unit-test meta-evaluation suite for judge failures in grounded QA.","以 144 个单元测试审计 grounded QA judge 的校准与具体失效检测能力。"],"why":"It tests whether a judge catches concrete grounded-answer failures rather than merely correlating with GPT-4.","primary_link":"https://aclanthology.org/2025.coling-main.304/","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/2025.coling-main.304.pdf"}],"link_count":2,"sections":9},{"id":"gui-360-2025","title":"GUI-360°: A Comprehensive Dataset and Benchmark for Computer-Using Agents","year":2025,"venue":"arXiv preprint; submitted to ICLR 2026 (no acceptance decision verified)","authors":["Jian Mu","Chaoyun Zhang","Chiming Ni","Lu Wang","Bo Qiao","Kartik Mathur","Qianhui Wu","Yuhang Xie","Xiaojun Ma","Mengyu Zhou","Si Qin","Liqun Li","Yu Kang","Minghua Ma","Qingwei Lin","Saravan Rajmohan","Dongmei Zhang"],"authors_zh":"Jian Mu, Chaoyun Zhang, Chiming Ni, Lu Wang, Bo Qiao, Kartik Mathur, Qianhui Wu, Yuhang Xie, Xiaojun Ma, Mengyu Zhou, Si Qin, Liqun Li, Yu Kang, Minghua Ma, Qingwei Lin, Saravan Rajmohan, Dongmei Zhang","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","benchmark","construction_recipe","model_report"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","desktop_gui_control","windows_office","gui_grounding","screen_parsing","action_prediction"],"tags":["gui-360","desktop-gui-agent","windows-office","agent-trajectory","multimodal-trajectory","hybrid-gui-api","llm-as-judge","failure-trajectories","sft","benchmark","environment-replay-incomplete","release-version-drift"],"status":"partial","priority":"必读","paper_type_zh":"Windows Office 智能体轨迹数据集、基准与构造报告","best_for_zh":"研究环境与智能体轨迹数据、GUI agent SFT、静态评测、failure retention 与 replay 审计的读者","confidence":"medium","one_line":["GUI-360° releases 17,189 successful and 62,170 failed Windows Office trajectories with screenshots, accessibility state, reasoning, and GUI/API actions, then derives grounding, screen-parsing, and action-prediction SFT/benchmark data; the collector, replay environment, judge prompt, and claimed RL setup remain incomplete.","GUI-360° 发布约 574 GB 的 Windows Office 多模态 GUI/API 轨迹、独立保留的失败轨迹及四类 SFT-ready 数据，但未发布 TrajAgent/EvaAgent 采集栈、VM replay、judge prompt 或可执行 terminal checker。"],"why":"It makes the full environment-data chain unusually visible - real-query sourcing, template instantiation, executed multimodal episodes, success/failure retention, whole-trajectory judgment, and step-level training views - while showing why static trajectory availability is not the same as replayable agent training or a verified RL feedback contract.","primary_link":"https://arxiv.org/abs/2511.04307","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/2020-qqtcg/GUI-360"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/vyokky/GUI-360"}],"link_count":5,"sections":9},{"id":"hallucounter-reference-free-hallucination-detection-2025","title":"HalluCounter: Reference-free LLM Hallucination Detection in the Wild!","year":2025,"venue":"Findings of IJCNLP-AACL 2025","authors":["Ashok Urlana","Gopichand Kanumolu","Charaka Vinayak Kumar","Bala Mallikarjunarao Garlapati","Rahul Mishra"],"authors_zh":"Ashok Urlana、Gopichand Kanumolu、Charaka Vinayak Kumar、Bala Mallikarjunarao Garlapati、Rahul Mishra","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["evaluation","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","summarization"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"可读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["HalluCounterEval is the paper's released feedback or evaluation resource.","HalluCounterEval 发布合成与人工专家标注的问答响应，并保留幻觉二元标记、置信度和最优回答，适合黑盒模型输出质量监督。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2025.findings-ijcnlp.20/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ashokurlana/HalluCounterEval"}],"link_count":3,"sections":9},{"id":"halogen-llm-hallucinations-2025","title":"HALoGEN: Fantastic LLM Hallucinations and Where to Find Them","year":2025,"venue":"ACL 2025","authors":["Abhilasha Ravichander","Shrusti Ghela","David Wadden","Yejin Choi"],"authors_zh":"Abhilasha Ravichander、Shrusti Ghela、David Wadden、Yejin Choi","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","summarization"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"必读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["HALoGEN is the paper's released feedback or evaluation resource.","以九类真实生成场景、10,923 提示和原子事实核验器标注 150K 输出，并按幻觉来源给出可解释 taxonomy，适合通用事实性研究。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2025.acl-long.71/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lasha-nlp/HALoGEN-prompts"},{"key":"project","label":["Project","项目主页"],"url":"https://halogen-hallucinations.github.io/"}],"link_count":4,"sections":9},{"id":"reasoning-economy-survey-2025","title":"Harnessing the Reasoning Economy: A Survey of Efficient Reasoning for Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Rui Wang","Hongru Wang","Boyang Xue","Jianhui Pang","Shudong Liu","Yi Chen","Jiahao Qiu","Derek Fai Wong","Heng Ji","Kam-Fai Wong"],"authors_zh":"Rui Wang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["test_time_compute","evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","chain-of-thought","reasoning-data","inference-efficiency","test-time-compute"],"tags":["foundations-and-primers","reasoning-economy","chain-of-thought","survey",2025],"status":"verified","priority":"必读","paper_type_zh":"高效推理综述","best_for_zh":"研究长推理过程、测试时预算或推理效率的读者。","confidence":"high","one_line":["A 2025 survey of how to reduce redundant reasoning without confusing fewer tokens with better reasoning.","讨论如何减少冗长推理，同时不把少输出误当成更会推理的 2025 综述。"],"why":"It gives students a disciplined way to compare answer quality against trace length, time, and computation.","primary_link":"https://arxiv.org/abs/2503.24377","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DevoAllen/Awesome-Reasoning-Economy-Papers"}],"link_count":3,"sections":9},{"id":"helpsteer2-preference-complementing-ratings-with-preferences-2025","title":"HelpSteer2-Preference: Complementing Ratings with Preferences","year":2025,"venue":"ICLR 2025","authors":["Zhilin Wang","Alexander Bukharin","Olivier Delalleau","Daniel Egert et al."],"authors_zh":"Zhilin Wang、Alexander Bukharin、Olivier Delalleau、Daniel Egert 等","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["preference-feedback-batch-2026","post-training","data-construction"],"status":"verified","priority":"必读","paper_type_zh":"偏好与奖励反馈数据集或数据构建研究","best_for_zh":"需要构建、审计或复用偏好与奖励反馈数据的研究者。","confidence":"high","one_line":["HelpSteer2-Preference: Complementing Ratings with Preferences contributes a preference/reward feedback data object or construction method.","将 HelpSteer2 的多维评分补充为成对偏好数据，并比较评分监督与偏好监督对奖励模型训练的作用。"],"why":"It exposes a reusable preference or reward-feedback data surface that requires provenance and bias audit before reuse.","primary_link":"https://arxiv.org/abs/2410.01257","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA/NeMo-Aligner"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/HelpSteer2"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/nvidia/Llama-3.1-Nemotron-70B-Reward"}],"link_count":4,"sections":9},{"id":"helpsteer3-preference-open-human-annotated-preference-data-across-diverse-tasks-and-languages-2025","title":"HelpSteer3-Preference: Open Human-Annotated Preference Data across Diverse Tasks and Languages","year":2025,"venue":"arXiv","authors":["Zhilin Wang","Jiaqi Zeng","Olivier Delalleau","Hoo-Chang Shin","Felipe Soares","Alexander Bukharin","Ellie Evans","Yi Dong","Oleksii Kuchaiev"],"authors_zh":"Zhilin Wang、Jiaqi Zeng、Olivier Delalleau 等（NVIDIA）","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["preference_learning","reward_modeling","safety_alignment","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"专家人工偏好数据、奖励模型与 RLHF","best_for_zh":"需要审计 LLM 后训练反馈、奖励或评测数据的研究者。","confidence":"high","one_line":["HelpSteer3-Preference is a CC-BY-4.0 human-annotated, multilingual and specialist preference dataset for reward modeling and RLHF.","发布 HelpSteer3-Preference 多任务、多语言人工偏好数据，扩展通用、STEM 与中文场景的奖励模型监督覆盖。"],"why":"It specifies specialist recruitment, multi-annotator preference strength, disagreement filtering, multilingual coverage, and the release boundary needed to reuse preference data responsibly.","primary_link":"https://arxiv.org/abs/2505.11475","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/HelpSteer3#preference"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/adlr/HelpSteer/"}],"link_count":3,"sections":9},{"id":"herald-lean4-dataset-2025","title":"Herald: A Natural Language Annotated Lean 4 Dataset","year":2025,"venue":"ICLR 2025","authors":["Guoxiong Gao","Yutong Wang","Jiedong Jiang","Qi Gao","Zihan Qin","Tianyi Xu","Bin Dong"],"authors_zh":"Guoxiong Gao、Yutong Wang、Jiedong Jiang、Qi Gao、Zihan Qin、Tianyi Xu、Bin Dong","tracks":["data_construction_open_release_recipes","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","model_report","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report"],"domains":["mathematics","formal_theorem_proving","lean4"],"tags":["herald","lean4","mathlib4","formal-mathematics","autoformalization","nl-fl-pairs","proof-informalization","tactic-state-augmentation","retrieval-augmented-generation","compiler-validation"],"status":"partial","priority":"必读","paper_type_zh":"Lean 4 自然语言—形式语言数据发布、构造 recipe 与模型报告","best_for_zh":"研究形式推理数据构造、autoformalization、tactic-state augmentation、混合验证契约，或审计形式有效性与语义对齐之间边界的读者","confidence":"high","one_line":["Herald dependency-orders Mathlib informalization, retrieves from 1,000 manual examples, and augments statements through tactic states and LLM rewrites, but omits the construction code, exact source revision, row lineage, rejects, and decontamination evidence.","Herald 按依赖层级对 Mathlib4 进行非形式化，检索 1,000 个人工示例，并通过 tactic state 与 LLM 改写扩充数据；它公开 579,883 条 statement rows 和 44,553 条 proof rows，但未公开完整构造代码、精确源版本、逐行 lineage、rejects 与去污染证据。"],"why":"It demonstrates how formal libraries can seed large parallel reasoning corpora while showing why compiler success must not be conflated with natural-language semantic correctness, proof availability, data quality, or complete release lineage.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/file/8c2bb821410066459be64d03a4dc5719-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/frenzymath/herald_translator"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/FrenzyMath/Herald_statements"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/FrenzyMath/Herald_translator"}],"link_count":12,"sections":9},{"id":"hal-2025","title":"Holistic Agent Leaderboard: The Missing Infrastructure for AI Agent Evaluation","year":2025,"venue":"ICLR 2026 Poster","authors":["Sayash Kapoor","Benedikt Stroebl","Peter Kirgis","Nitya Nadgir","Zachary S Siegel","Boyi Wei","Tianci Xue","Ziru Chen","Felix Chen","Saiteja Utpala","Franck Ndzomga","Dheeraj Oruganty","Sophie Luskin","Kangheng Liu","Botao Yu","Amit Arora","Dongyoon Hahm","Harsh Trivedi","Huan Sun","Juyong Lee","Tengjun Jin","Yifan Mai","Yifei Zhou","Yuxuan Zhu","Rishi Bommasani","Daniel Kang","Dawn Song","Peter Henderson","Yu Su","Percy Liang","Arvind Narayanan"],"authors_zh":"Sayash Kapoor、Benedikt Stroebl、Peter Kirgis、Nitya Nadgir、Zachary S Siegel、Boyi Wei、Tianci Xue、Ziru Chen、Felix Chen、Saiteja Utpala、Franck Ndzomga、Dheeraj Oruganty、Sophie Luskin、Kangheng Liu、Botao Yu、Amit Arora、Dongyoon Hahm、Harsh Trivedi、Huan Sun、Juyong Lee、Tengjun Jin、Yifan Mai、Yifei Zhou、Yuxuan Zhu、Rishi Bommasani、Daniel Kang、Dawn Song、Peter Henderson、Yu Su、Percy Liang、Arvind Narayanan","tracks":["environment_agent_trajectory_data"],"source_role":["infrastructure","benchmark","agent_environment","data_release","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["trace_writing","reward_verifier_layer","scaling_report","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","web_navigation","software_engineering","scientific_research","customer_service","cost_aware_evaluation","behavioral_audit"],"tags":["environment-agent-trajectory-data","agent-evaluation","agent-trajectories","tool-use","evaluation-harness","leaderboard","cost-aware-evaluation","behavioral-audit","benchmark-gaming","contamination","replay-risk","version-drift"],"status":"partial","priority":"必读","paper_type_zh":"智能体评测基础设施、轨迹发布与行为审计研究","best_for_zh":"研究 environment/agent trajectory data、混合 evaluator、benchmark contamination、版本漂移与 replay 审计的读者","confidence":"medium","one_line":["HAL standardizes nine agent benchmarks into cost-aware run archives with task evaluator outputs and Weave call logs; the paper reports 21,730 rollouts/2.5B tokens, but the mutable 113 GB release lacks a dataset card, license, and frozen paper manifest.","HAL 将九个异构 agent benchmark 统一为带任务 evaluator、成本与 Weave 调用轨迹的评测运行；但论文的 21,730 个 rollout、项目当前的 26,597 个 rollout 与 HF 当前 380 个加密 run archive 不能混为同一语料快照。"],"why":"HAL shows what an auditable agent-trajectory corpus needs beyond final accuracy: task-linked action/call traces, environment and evaluator outputs, cost, errors, success/failure retention, and run lineage. It also demonstrates that public logs are not automatically training-ready when rights, privacy, replay, judge calibration, and paper-snapshot provenance remain unresolved.","primary_link":"https://openreview.net/forum?id=vUaY1t64ZZ","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/princeton-pli/hal-harness"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/agent-evals/hal_traces"},{"key":"project","label":["Project","项目主页"],"url":"https://hal.cs.princeton.edu/"}],"link_count":8,"sections":9},{"id":"kernel-divergence-score-contamination-2025","title":"How Contaminated Is Your Benchmark? Measuring Dataset Leakage in Large Language Models with Kernel Divergence","year":2025,"venue":"ICML 2025","authors":["Hyeong Kyu Choi","Maxim Khanov","Hongxin Wei","Yixuan Li"],"authors_zh":"Hyeong Kyu Choi, Maxim Khanov, Hongxin Wei, Yixuan Li","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["contamination","verifier-audit","evaluation-reliability"],"tags":["kernel-divergence","audit","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"需要复盘基准污染、评估偏差、验证器失效或奖励投机风险的研究者。","confidence":"high","one_line":["Kernel Divergence Score compares embedding-kernel structure before and after benchmark fine-tuning to quantify leakage.","以微调前后样本嵌入核矩阵的结构变化，量化候选基准的数据集级泄漏。"],"why":"It exposes leakage as an empirical precondition of a benchmark result instead of assuming scores are uncontaminated.","primary_link":"https://proceedings.mlr.press/v267/choi25b.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deeplearning-wisc/kernel-divergence-score"}],"link_count":3,"sections":9},{"id":"forgetting-data-contamination-2025","title":"How Much Can We Forget about Data Contamination?","year":2025,"venue":"ICML 2025","authors":["Sebastian Bordt","Suraj Srinivas","Valentyn Boreiko","Ulrike von Luxburg"],"authors_zh":"Sebastian Bordt, Suraj Srinivas, Valentyn Boreiko, Ulrike von Luxburg","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["contamination","verifier-audit","evaluation-reliability"],"tags":["audit","data-contamination","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"基准污染与训练遗忘的受控实验研究","best_for_zh":"需要量化训练数据污染对基准评测影响的研究者。","confidence":"high","one_line":["Controlled training experiments measure when repeated benchmark exposure is forgotten as training data scales.","基准污染的影响取决于重复次数、模型与训练规模；大量后续新数据可使早期污染痕迹自然遗忘。"],"why":"It replaces an all-or-nothing leakage assumption with a measurable interaction among repetition, training scale, and weight decay.","primary_link":"https://proceedings.mlr.press/v267/bordt25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tml-tuebingen/forgetting-contamination"}],"link_count":2,"sections":9},{"id":"multilingual-judge-reliability-2025","title":"How Reliable is Multilingual LLM-as-a-Judge?","year":2025,"venue":"Findings of EMNLP 2025","authors":["Xiyan Fu","Wei Liu"],"authors_zh":"Xiyan Fu, Wei Liu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Accepted multilingual judge-reliability study with public ACL paper materials.","审计多语 LLM 评测器的跨语言一致性，发现平均 Fleiss' Kappa 约为 0.3。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.findings-emnlp.587/","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/2025.findings-emnlp.587.pdf"}],"link_count":2,"sections":9},{"id":"how-to-evaluate-reward-models-rlhf-2025","title":"How to Evaluate Reward Models for RLHF","year":2025,"venue":"ICLR 2025","authors":["Evan Frick","Tianle Li","Connor Chen","Wei-Lin Chiang","Anastasios N. Angelopoulos","Jiantao Jiao","Banghua Zhu","Joseph E. Gonzalez","Ion Stoica"],"authors_zh":"Evan Frick, Tianle Li, Connor Chen, Wei-Lin Chiang, Anastasios N. Angelopoulos, Jiantao Jiao, Banghua Zhu, Joseph E. Gonzalez, Ion Stoica","tracks":["preference_reward_feedback_data","audit_failure_contamination_verifier_attacks"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","rlhf","evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"奖励模型后效预测评测与偏好代理数据论文","best_for_zh":"需要以较低成本预测奖励模型是否能带来真实 RLHF 下游人类偏好改善的研究者。","confidence":"high","one_line":["PPE evaluates reward models on human-preference and correctness proxies calibrated against real post-RLHF preference outcomes.","PPE 将 12 个领域的人类偏好与可验证正确性代理指标连接到真实 RLHF 后效，适合检验奖励数据是否真正有训练价值。"],"why":"It provides an evidence-backed way to test whether an offline reward-model score predicts the effect that matters: downstream RLHF quality.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/2e01083b381b4865919b4915ef32e3d2-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lmarena/PPE"}],"link_count":5,"sections":9},{"id":"hunyuan-a13b-technical-report-2025","title":"Hunyuan-A13B Technical Report","year":2025,"venue":"Technical report","authors":["Tencent Hunyuan Team"],"authors_zh":"Tencent Hunyuan Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","science","technology","engineering"],"tags":["hunyuan-a13b","technical-report","grpo","grm","sandbox","synthetic-data","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿模型技术报告与数据披露账本","best_for_zh":"需要审计大规模 STEM 数据、GRPO、sandbox/GRM 反馈及后训练可复现性边界的读者","confidence":"medium","one_line":["A frontier-model disclosure describing a >20T-token foundation corpus, a 250B-token STEM subset, four post-training stages, a 150K-problem reasoning-RL mixture, a five-role agent synthesis engine, and a multi-service reward stack, while leaving record-level data and rollout artefacts unreleased.","Hunyuan-A13B 报告了 20T 级数据、250B STEM 数据、四个 SFT/RL 阶段、GRPO 及 sandbox/GRM 反馈；但来源清单、教师模型、奖励权重和审计工件仍未披露或未核验。"],"why":"For the frontier-reports data-disclosure track, Hunyuan-A13B provides unusually concrete stage-level counts and verifier/environment descriptions that can be separated into auditable claims. It is still a model report rather than a released dataset: exact records, provenance, splits, licences, verifier implementations, and RL trajectories remain unavailable.","primary_link":"https://github.com/Tencent-Hunyuan/Hunyuan-A13B/blob/main/report/Hunyuan_A13B_Technical_Report.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent-Hunyuan/Hunyuan-A13B"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/tencent/Hunyuan-A13B-Instruct"}],"link_count":3,"sections":9},{"id":"hunyuan-mt-2025","title":"Hunyuan-MT Technical Report","year":2025,"venue":"arXiv preprint","authors":["Mao Zheng","Zheng Li","Bingxin Qu","Mingyang Song","Yang Du","Mingrui Sun","Di Wang"],"authors_zh":"Mao Zheng 等（Tencent Hunyuan）","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","process_supervision","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward","process_reward"],"training_use":["sft","distillation","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["machine_translation","multilingual_translation","low_resource_languages","minority_languages","dialects","terminology","chinese_translation"],"tags":["hunyuan-mt","tencent-hunyuan","machine-translation","multilingual","low-resource-languages","minority-languages","synthetic-translation","deepseek-v3-teacher","gemba","xcomet","cometkiwi","terminology-reward","grpo","weak-to-strong","chimera","six-candidate-fusion","test-time-scaling","process-reward","benchmark-contamination-risk","restrictive-open-weight-license","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"多语机器翻译数据构造与强化学习技术报告","best_for_zh":"研究多语语料配比、翻译 SFT/GRPO、弱到强融合、少数民族语言数据权利与评测污染的读者","confidence":"high","one_line":["Hunyuan-MT maps a reported 1.3T multilingual foundation through RegMix-style CPT, a 3M-to-268K SFT funnel, judge-based GRPO, and six-candidate Chimera fusion, but releases no training manifests, rewards, rollouts, splits, or paper-run lineage.","Hunyuan-MT 将论文所述 1.3T 多语基础，经 RegMix-style CPT、3M→268K SFT 漏斗、judge-based GRPO 与六候选 Chimera fusion 串成完整路线，但未发布训练 manifest、reward、rollout、split 或 paper-run lineage。"],"why":"It exposes a concrete translation lifecycle while making public-test reuse, correlated teacher/judge feedback, minority-language rights, test-time budgets, and restrictive licensing auditable.","primary_link":"https://arxiv.org/abs/2509.05209","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent-Hunyuan/Hunyuan-MT"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/tencent/hunyuan-mt-68b42f76d473f82798882597"},{"key":"project","label":["Project","项目主页"],"url":"https://hunyuan.tencent.com/"}],"link_count":5,"sections":9},{"id":"hunyuan-t1-2025","title":"Hunyuan-T1","year":2025,"venue":"official release page","authors":["Tencent Hunyuan Team"],"authors_zh":"Tencent Hunyuan Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","logic","science","code"],"tags":["frontier-report","hunyuan","reinforcement-learning","curriculum-learning","data-replay","policy-reset","self-reward","reward-model","disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿推理模型官方发布","best_for_zh":"研究 RL、奖励披露与推理模型系统报告的读者","confidence":"medium","one_line":["Hunyuan-T1 reports that 96.7% of post-training compute went to RL over broad science-and-reasoning problems, with curriculum, replay/reset, ground-truth feedback, and T1-preview-plus-reward-model preference scoring, but releases none of the underlying training records or reward implementation.","Hunyuan-T1 披露了重 RL 的后训练、课程和混合奖励，但未公开数据与奖励账本。"],"why":"It is a compact disclosure-ledger case: several system-level post-training components are named, yet the data object, verifier semantics, reward aggregation, stage accounting, rights, and reproducibility artifacts remain unknown.","primary_link":"https://tencent.github.io/llm.hunyuan.T1/README_EN.html","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/Tencent/llm.hunyuan.T1"}],"link_count":2,"sections":9},{"id":"hunyuan-turbos-2025","title":"Hunyuan-TurboS: Advancing Large Language Models through Mamba-Transformer Synergy and Adaptive Chain-of-Thought","year":2025,"venue":"arXiv preprint","authors":["Tencent Hunyuan Team"],"authors_zh":"Tencent Hunyuan Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","code","science","logic","multilingual","safety","finance","legal","medical"],"tags":["hunyuan-turbos","tencent-hunyuan","frontier-report","data-disclosure-ledger","adaptive-cot","sft","grm","grpo","reward-system","code-sandbox","hosted-model"],"status":"partial","priority":"可读","paper_type_zh":"前沿推理模型技术报告与数据披露账本","best_for_zh":"需要审计推理后训练数据、反馈组件与披露边界的读者","confidence":"medium","one_line":["Hunyuan-TurboS reports 16T pre-training tokens, 3M SFT samples, adaptive teacher-generated traces, about 200K human preference labels, more than 800K executable code samples, and 300K plus 160K two-stage GRPO inputs, but releases no training corpus, weights, construction code, reward checkpoints, or replay ledger.","Hunyuan-TurboS 报告 300 万条 SFT 指令、自适应长短 CoT、人工标注 GRM、代码沙箱反馈和两阶段 GRPO，但未公开来源 manifest 与多数关键复现配置。"],"why":"It shows how a frontier report can disclose substantial stage-level data and feedback contracts while a hosted model demo and report repository remain fundamentally different from open weights, reusable data, and a reproducible training pipeline.","primary_link":"https://arxiv.org/abs/2505.15431","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/spaces/tencent/hunyuan-turbos"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/Tencent-Hunyuan/Hunyuan-TurboS"}],"link_count":5,"sections":9},{"id":"hybrid-preferences-multipref-2025","title":"Hybrid Preferences: Learning to Route Instances for Human vs. AI Feedback","year":2025,"venue":"ACL 2025","authors":["Lester James V. Miranda","Yizhong Wang","Yanai Elazar","Sachin Kumar","Valentina Pyatkin","Faeze Brahman","Noah A. Smith","Hannaneh Hajishirzi","Pradeep Dasigi"],"authors_zh":"Lester James V. Miranda、Yizhong Wang、Yanai Elazar、Sachin Kumar、Valentina Pyatkin、Faeze Brahman、Noah A. Smith、Hannaneh Hajishirzi、Pradeep Dasigi","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["Hybrid Preferences releases MultiPref, where the same response comparisons carry crowdworker, expert, and language-model feedback for studying when human annotation is worth its additional cost.","Hybrid Preferences 发布 MultiPref：同一回答比较同时带有众包、专家和语言模型反馈，用于研究何时值得为人工标注付出额外成本。"],"why":"MultiPref pairs the same response comparisons with crowdworker, expert, and LM annotations for feedback-routing research.","primary_link":"https://aclanthology.org/2025.acl-long.355/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/hybrid-preferences"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/multipref"}],"link_count":3,"sections":9},{"id":"icon2-self-synthetic-preference-2025","title":"ICON2: Aligning Large Language Models Using Self-Synthetic Preference Data via Inherent Regulation","year":2025,"venue":"EMNLP 2025","authors":["Qiyuan Chen","Hongsen Huang","Qian Shao","Jiahe Chen","Jintai Chen","Hongxia Xu","Renjie Hua","Chuan Ren","Jian Wu"],"authors_zh":"Qiyuan Chen、Hongsen Huang、Qian Shao、Jiahe Chen、Jintai Chen、Hongxia Xu、Renjie Hua、Chuan Ren、Jian Wu（浙江大学、苏州证券、微医云等）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["trace_writing"],"domains":["alignment","instruction-following"],"tags":["synthetic-preference-data","representation-steering","instruction-selection","direct-preference-optimization"],"status":"verified","priority":"必读","paper_type_zh":"自合成偏好数据构建与直接偏好优化研究","best_for_zh":"适合构建目标模型特定偏好语料、又不希望反复采样或依赖外部评审的读者。","confidence":"high","one_line":["ICON2 filters self-synthesized instructions by representation consistency and directly generates chosen and rejected responses through opposite internal steering directions.","ICON2 以表征一致性筛选自合成指令，再用相反的内部引导方向直接生成优选和拒选回答。"],"why":"It makes a target model's representation space an explicit source of both data-selection and pair-construction decisions.","primary_link":"https://aclanthology.org/2025.emnlp-main.196/","links":[],"link_count":2,"sections":9},{"id":"impossiblebench-measuring-llms-propensity-of-exploiting-test-cases","title":"ImpossibleBench: Measuring LLMs' Propensity of Exploiting Test Cases","year":2025,"venue":"arXiv","authors":["Ziqian Zhong","Aditi Raghunathan","Nicholas Carlini"],"authors_zh":"Ziqian Zhong, Aditi Raghunathan, Nicholas Carlini","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Conflicting specifications and tests make every pass a deterministic cheating label.","通过构造规格与测试冲突的任务，让任何 pass 都成为确定性作弊标签。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2510.20270","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/safety-research/impossiblebench"}],"link_count":2,"sections":9},{"id":"improve-judge-ability-2025","title":"Improve LLM-as-a-Judge Ability as a General Ability","year":2025,"venue":"EMNLP 2025","authors":["Jiachen Yu","Shaoning Sun","Xiaohui Hu","Jiaxu Yan","Kaidong Yu","Xuelong Li"],"authors_zh":"Jiachen Yu, Shaoning Sun, Xiaohui Hu, Jiaxu Yan, Kaidong Yu, Xuelong Li","tracks":["audit_failure_contamination_verifier_attacks","preference_reward_feedback_data"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"生成式 LLM 裁判的训练、偏差过滤与偏好信号评测","best_for_zh":"构建开源 LLM 裁判，或用自动偏好信号训练策略模型的团队。","confidence":"medium","one_line":["Two-stage SFT-DPO training turns filtered pairwise judgments into an open generative judge and tests its downstream preference signals.","以经双重核验的成对判决数据进行 SFT-DPO 训练，并审计其作为偏好信号的效用。"],"why":"It exposes the data filters and evaluation boundary behind an open generative judge used as an RLAIF signal.","primary_link":"https://aclanthology.org/2025.emnlp-main.712/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/R-I-S-E/RISE-Judge-SFT-20K"}],"link_count":4,"sections":9},{"id":"redis-reasoning-distillation-2025","title":"Improving In-Context Learning with Reasoning Distillation","year":2025,"venue":"arXiv preprint (under review)","authors":["Nafis Sadeq","Xin Xu","Zhouhang Xie","Julian McAuley","Byungkyu Kang","Prarit Lamba","Xiang Gao"],"authors_zh":"Nafis Sadeq, Xin Xu, Zhouhang Xie, Julian McAuley, Byungkyu Kang, Prarit Lamba, Xiang Gao","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","reasoning-data"],"status":"verified","priority":"可读","paper_type_zh":"推理数据构造与监督微调论文","best_for_zh":"研究推理记录如何进入后训练的读者。","confidence":"high","one_line":["ReDis distils rule generation and rule following into smaller models for inductive in-context learning.","ReDis 将规则生成和规则遵循蒸馏到较小模型中，以改善归纳式上下文学习。"],"why":"It makes a reasoning-data consumption decision explicit.","primary_link":"https://arxiv.org/abs/2504.10647","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NafisSadeq/reasoning-distillation"}],"link_count":2,"sections":9},{"id":"cdg-self-playing-game-2025","title":"Improving Rationality in the Reasoning Process of Language Models through Self-playing Game","year":2025,"venue":"ICML 2025","authors":["Pinzheng Wang","Juntao Li","Zecheng Tang","Haijia Gui","Min Zhang"],"authors_zh":"Pinzheng Wang、Juntao Li、Zecheng Tang、Haijia Gui、Min Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["self_play_anchor","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","logical_reasoning","scientific_reasoning"],"tags":["critic-discernment-game","self-play","critique-feedback","reasoning-revision","adversarial-critic","rest"],"status":"verified","priority":"必读","paper_type_zh":"批评辨别自博弈数据与反馈训练配方","best_for_zh":"希望复用自博弈推理反馈数据、核查角色奖励与发布边界，或研究批评采纳与拒绝机制的读者","confidence":"high","one_line":["CDG generates one-step Prover–Critic–revision episodes and converts correctness- thresholded helpful, misleading, corrected, and resistant outcomes into role-specific training data.","CDG 生成一次 Prover—Critic—修订回合，并把通过正确性阈值筛选的纠错、误导、修正与抗误导结果转成分角色训练数据。"],"why":"It provides a concrete open recipe for constructing reasoning-feedback data while exposing how answer-only verification, exact-phrase rewards, correlated self-play, and incomplete release lineage can shape the resulting supervision.","primary_link":"https://proceedings.mlr.press/v267/wang25bb.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PinzhengWang322/Critic_Discernment_Game"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/drive/folders/1OBv9Gpk_Hrl4BywhQ6V295SKzQJk3ZoN?usp=sharing"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/PinzhengWang/Llama_CDG"}],"link_count":7,"sections":9},{"id":"ineq-comp-benchmarking-human-intuitive-compositional-reasoning-in-automated-theorem-prov","title":"Ineq-Comp: Benchmarking Human-Intuitive Compositional Reasoning in Automated Theorem Proving on Inequalities","year":2025,"venue":"NeurIPS 2025","authors":["Haoyu Zhao","Yihan Geng","Shange Tang","Yong Lin","Bohan Lyu","Hongzhou Lin","Chi Jin","Sanjeev Arora"],"authors_zh":"Haoyu Zhao, Yihan Geng, Shange Tang, Yong Lin, Bohan Lyu, Hongzhou Lin, Chi Jin, Sanjeev Arora","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A 375-task Lean benchmark created by controlled transformations of elementary inequalities to test composition.","375 个从基础不等式受控变换得到的 Lean 任务，专门测组合泛化。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2505.12680","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/haoyuzhao123/LeanIneqComp"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zzzzzhy/Ineq-Comp"}],"link_count":4,"sections":9},{"id":"deepseek-grm-inference-time-scaling-2025","title":"Inference-Time Scaling for Generalist Reward Modeling","year":2025,"venue":"arXiv preprint","authors":["Zijun Liu","Peiyi Wang","Runxin Xu","Shirong Ma","Chong Ruan","Peng Li","Yang Liu","Yu Wu"],"authors_zh":"Zijun Liu；Peiyi Wang；Runxin Xu；Shirong Ma；Chong Ruan；Peng Li；Yang Liu；Yu Wu","tracks":["rollout_search_test_time_trace_data"],"source_role":["verifier_reward","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["process_reward","scalar_reward"],"training_use":["reward_modeling","evaluation","test_time_compute"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["reasoning"],"tags":["reward-model","generalist-reward-model","principles","critiques","inference-time-scaling"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A generalist reward-model pipeline whose inference-time scaling relies on generated principles and critiques.","该工作以 SPCT 让通用生成式奖励模型在线生成原则和批评，再以并行采样和 Meta RM 投票扩展奖励判断的测试时计算。"],"why":"It belongs in Track 5 as verifier infrastructure, while its closed data lineage makes it unsuitable to represent as an open trace corpus.","primary_link":"https://arxiv.org/abs/2504.02495","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/BBQGOD/deepseek-grm-68b4681169dbb97fd30614b5"}],"link_count":2,"sections":9},{"id":"classical-search-diffusion-tts-2025","title":"Inference-time Scaling of Diffusion Models through Classical Search","year":2025,"venue":"arXiv preprint","authors":["Xiangcheng Zhang","Haowei Lin","Haotian Ye","James Zou","Jianzhu Ma","Yitao Liang","Yilun Du"],"authors_zh":"Xiangcheng Zhang、Haowei Lin、Jianzhu Ma、Yitao Liang（Helixon）；Haotian Ye、James Zou（斯坦福大学）；Yilun Du（哈佛大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-search"],"tags":["test-time-compute","diffusion","search","verifier"],"status":"verified","priority":"可读","paper_type_zh":"扩散模型测试时搜索研究","best_for_zh":"研究非自回归测试时扩展的读者。","confidence":"high","one_line":["Classical search improves diffusion inference through verifier-guided Langevin local search and BFS/DFS global exploration.","该工作以朗之万局部搜索和 BFS/DFS 全局搜索扩展扩散模型推理。"],"why":"It makes local refinement and global branching explicit compute-allocation choices.","primary_link":"https://arxiv.org/abs/2505.23614","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/helixml/diffusion-inference-scaling"},{"key":"project","label":["Project","项目主页"],"url":"https://diffusion-inference-scaling.github.io/"}],"link_count":4,"sections":9},{"id":"insta-internet-scale-agents-2025","title":"InSTA: Towards Internet-Scale Training For Agents","year":2025,"venue":"arXiv preprint; submitted to ICLR 2026 (no acceptance record found)","authors":["Brandon Trabucco","Gunnar Sigurdsson","Robinson Piramuthu","Ruslan Salakhutdinov"],"authors_zh":"Brandon Trabucco、Gunnar Sigurdsson、Robinson Piramuthu、Ruslan Salakhutdinov","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","agent_environment","verifier_reward","model_report"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["web_agents","gui_agents","browser_automation","multimodal_reasoning"],"tags":["insta","insta-150k","web-agents","browser-automation","playwright","common-crawl","task-generation","trajectory-generation","llm-judge","success-filtering","qwen3","sft","live-environment-drift","failed-trajectory-gap","release-completeness","privacy-risk","copyright-risk","count-reconciliation"],"status":"partial","priority":"必读","paper_type_zh":"Internet-scale web-agent task/trajectory construction recipe、live-browser environment、LLM-judge filtering 与 task-data release","best_for_zh":"研究 web-agent SFT、live-site task generation、Playwright rollout、learned verifier，或审计动态网站数据隐私、安全、版权和发布完整性的读者","confidence":"high","one_line":["InSTA combines an LLM safety and task proposer, one live Playwright exploration loop, LLM-agent rollouts, and LLM success judgment to construct internet-scale web-agent SFT data.","InSTA 用 LLM 对 100 万个 Common Crawl 排名站点进行安全筛选与任务生成，再以 live Playwright rollout 和 LLM success judge 构造 web-agent SFT 数据；论文与 v2 的精确任务数为 146,746，但当前官方发布只有约 146K 条 task rows 与 recipe code，没有论文声称的完整 multimodal trajectories、失败样本、judge rationales 或 checkpoints。"],"why":"It demonstrates a scalable path from ranked websites to environment-grounded agent training while making release completeness, learned-judge error, live-site drift, side effects, privacy defaults, source rights, and versioned count lineage first-class data-quality concerns.","primary_link":"https://arxiv.org/abs/2502.06776","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/data-for-agents/insta"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/data-for-agents/insta-150k-v3"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/data-for-agents"},{"key":"project","label":["Project","项目主页"],"url":"https://data-for-agents.github.io/"}],"link_count":13,"sections":9},{"id":"intellect-3-technical-report-2025","title":"INTELLECT-3: Technical Report","year":2025,"venue":"arXiv preprint","authors":["Prime Intellect Team","Mika Senghaas","Fares Obeid","Sami Jaghouar","William Brown","Jack Min Ong","Daniel Auras","Matej Sirovatka","Jannik Straube","Andrew Baker","Sebastian Müller","Justus Mattern","Manveer Basra","Aiman Ismail","Dominik Scherm","Cooper Miller","Ameen Patel","Simon Kirsten","Mario Sieg","Christian Reetz","Kemal Erdem","Vincent Weisser","Johannes Hagemann"],"authors_zh":"Prime Intellect Team 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["general_reasoning","mathematics","code","science","agent"],"tags":["frontier-report","data-disclosure-ledger","asynchronous-rl","verifiers","environments","agentic-rl","open-infrastructure"],"status":"partial","priority":"可读","paper_type_zh":"前沿模型技术报告与数据披露账本","best_for_zh":"关注开放 RL 基础设施、环境反馈、异步训练和数据可复现性的研究者","confidence":"high","one_line":["Prime Intellect reports SFT and asynchronous RL of a 106B MoE on released verifier-backed environments, with open infrastructure but no frozen run-level data manifest.","Prime Intellect 披露了 106B MoE 的 SFT 与大规模异步 RL 基础设施和环境，但未提供冻结的运行级数据清单。"],"why":"It lets reviewers separate genuinely open RL infrastructure and environments from undisclosed mixture weights, rollout settings, and record-level lineage.","primary_link":"https://arxiv.org/abs/2512.16144","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PrimeIntellect-ai/prime-rl"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/PrimeIntellect/INTELLECT-3"},{"key":"project","label":["Project","项目主页"],"url":"https://www.primeintellect.ai/blog/intellect-3"}],"link_count":5,"sections":9},{"id":"intermt-multiturn-interleaved-preference-2025","title":"InterMT: Multi-Turn Interleaved Preference Alignment with Human Feedback","year":2025,"venue":"arXiv preprint","authors":["Boyuan Chen","Donghai Hong","Jiaming Ji","Jiacheng Zheng","Bowen Dong","Jiayi Zhou","Kaile Wang","Juntao Dai","Xuyao Wang","Wenqi Chen","Qirui Zheng","Wenxin Li","Sirui Han","Yike Guo","Yaodong Yang"],"authors_zh":"Boyuan Chen、Donghai Hong、Jiaming Ji、Jiacheng Zheng、Bowen Dong、Jiayi Zhou、Kaile Wang、Juntao Dai、Xuyao Wang、Wenqi Chen、Qirui Zheng、Wenxin Li、Sirui Han、Yike Guo、Yaodong Yang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["InterMT releases human preference annotations for competing multi-turn, interleaved image-text conversations, covering both local turns and complete interaction trajectories.","InterMT 发布竞争性多轮图文交错对话的人工偏好标注，同时覆盖局部回合与完整交互轨迹。"],"why":"Human preferences over multi-turn interleaved multimodal conversations with local and global feedback.","primary_link":"https://arxiv.org/abs/2505.23950","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PKU-Alignment/InterMT"},{"key":"project","label":["Project","项目主页"],"url":"https://pku-intermt.github.io/"}],"link_count":3,"sections":9},{"id":"intern-s1-2025","title":"Intern-S1: A Scientific Multimodal Foundation Model","year":2025,"venue":"arXiv preprint","authors":["Lei Bai","Zhongrui Cai","Yuhang Cao","Maosong Cao","Weihan Cao","Chiyu Chen","Haojiong Chen","Kai Chen","Pengcheng Chen","Ying Chen","Yongkang Chen","Yu Cheng","Pei Chu","Tao Chu","Erfei Cui","Ganqu Cui","Long Cui","Ziyun Cui","Nianchen Deng","Ning Ding","Nanqing Dong","Peijie Dong","Shihan Dou","Sinan Du","Haodong Duan","Caihua Fan","Ben Gao","Changjiang Gao","Jianfei Gao","Songyang Gao","Yang Gao","Zhangwei Gao","Jiaye Ge","Qiming Ge","Lixin Gu","Yuzhe Gu","Aijia Guo","Qipeng Guo","Xu Guo","Conghui He","Junjun He","Yili Hong","Siyuan Hou","Caiyu Hu","Hanglei Hu","Jucheng Hu","Ming Hu","Zhouqi Hua","Haian Huang","Junhao Huang","Xu Huang","Zixian Huang","Zhe Jiang","Lingkai Kong","Linyang Li","Peiji Li","Pengze Li","Shuaibin Li","Tianbin Li","Wei Li","Yuqiang Li","Dahua Lin","Junyao Lin","Tianyi Lin","Zhishan Lin","Hongwei Liu","Jiangning Liu","Jiyao Liu","Junnan Liu","Kai Liu","Kaiwen Liu","Kuikun Liu","Shichun Liu","Shudong Liu","Wei Liu","Xinyao Liu","Yuhong Liu","Zhan Liu","Yinquan Lu","Haijun Lv","Hongxia Lv","Huijie Lv","Qitan Lv","Ying Lv","Chengqi Lyu","Chenglong Ma","Jianpeng Ma","Ren Ma","Runmin Ma","Runyuan Ma","Xinzhu Ma","Yichuan Ma","Zihan Ma","Sixuan Mi","Junzhi Ning","Wenchang Ning","Xinle Pang","Jiahui Peng","Runyu Peng","Yu Qiao","Jiantao Qiu","Xiaoye Qu","Yuan Qu","Yuchen Ren","Fukai Shang","Wenqi Shao","Junhao Shen","Shuaike Shen","Chunfeng Song","Demin Song","Diping Song","Chenlin Su","Weijie Su","Weigao Sun","Yu Sun","Qian Tan","Cheng Tang","Huanze Tang","Kexian Tang","Shixiang Tang","Jian Tong","Aoran Wang","Bin Wang","Dong Wang","Lintao Wang","Rui Wang","Weiyun Wang","Wenhai Wang","Jiaqi Wang","Yi Wang","Ziyi Wang","Ling-I Wu","Wen Wu","Yue Wu","Zijian Wu","Linchen Xiao","Shuhao Xing","Chao Xu","Huihui Xu","Jun Xu","Ruiliang Xu","Wanghan Xu","GanLin Yang","Yuming Yang","Haochen Ye","Jin Ye","Shenglong Ye","Jia Yu","Jiashuo Yu","Jing Yu","Fei Yuan","Yuhang Zang","Bo Zhang","Chao Zhang","Chen Zhang","Hongjie Zhang","Jin Zhang","Qiaosheng Zhang","Qiuyinzhe Zhang","Songyang Zhang","Taolin Zhang","Wenlong Zhang","Wenwei Zhang","Yechen Zhang","Ziyang Zhang","Haiteng Zhao","Qian Zhao","Xiangyu Zhao","Xiangyu Zhao","Bowen Zhou","Dongzhan Zhou","Peiheng Zhou","Yuhao Zhou","Yunhua Zhou","Dongsheng Zhu","Lin Zhu","Yicheng Zou"],"authors_zh":"Intern-S1 Team, Shanghai AI Laboratory","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["frontier_pipeline","reward_verifier_layer","optimizer_scaffold"],"domains":["general_reasoning","science"],"tags":["intern-s1","frontier-report","disclosure-ledger","scientific-data","multimodal-reasoning","continual-pretraining","instruction-data","reinforcement-learning","mixture-of-rewards","internbootcamp","compassverifier","polar-7b"],"status":"partial","priority":"可读","paper_type_zh":"前沿模型技术报告与数据披露账本","best_for_zh":"审计持续预训练、RL 任务和奖励披露边界的读者","confidence":"medium","one_line":["Intern-S1 reports 5T continual-pretraining tokens, best-of-N instruction curation, and 8-rollout online RL across more than 1,000 Internbootcamp tasks using CompassVerifier, rules, environment feedback, and POLAR-7B; weights are open, but training records and the full reward stack are not.","Intern-S1 报告了 5T 持续预训练 token（其中超过 2.5T 为科学 token），以及 InternBootCamp 中从离线到在线的 RL 与覆盖超过 1,000 个任务的 Mixture-of-Rewards；任务、数据和奖励细节仍为 unknown。"],"why":"It lets the frontier-disclosure track inspect how scientific sources, instruction records, synthetic tasks, verifiers, and open-ended rewards are combined, while keeping the boundary between a released checkpoint and an unreleased training pipeline explicit.","primary_link":"https://arxiv.org/abs/2508.15763","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InternLM/Intern-S1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/internlm/Intern-S1"}],"link_count":5,"sections":9},{"id":"microsoft-mai-ds-r1-2025","title":"Introducing MAI-DS-R1","year":2025,"venue":"Microsoft Azure AI Foundry technical release","authors":["Microsoft"],"authors_zh":"Microsoft","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","frontier_pipeline","release_audit"],"domains":["general","safety"],"tags":["microsoft","mai-ds-r1","deepseek-r1","frontier-report","data-disclosure-ledger","safety-alignment","multilingual","chain-of-thought","post-training"],"status":"partial","priority":"可读","paper_type_zh":"模型发布说明与数据披露账本","best_for_zh":"需要审计大厂后训练数据构造、合成 CoT 与安全数据边界的研究者","confidence":"high","one_line":["Microsoft's MAI-DS-R1 release reports about 350K keyword-derived multilingual questions with DeepSeek-R1/internal-model bootstrapped answers and CoT plus 110K named Tulu3 safety examples, but not the underlying records, feedback contract, splits, or source-level rights.","Microsoft 的 MAI-DS-R1 发布披露约 350K 个关键词派生的多语问题及由 DeepSeek-R1/内部模型引导的答案与 CoT，加上 110K 个具名 Tulu3 安全样本；底层记录、反馈合约、切分和来源级权利未披露。"],"why":"It is a concrete disclosure-limited example of a major company's post-training intervention: the reported construction stages and counts are useful, but the unreleased traces, internal teachers, reward details, and provenance block independent replication or safety-data audit.","primary_link":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-ds-r1/4405076","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/microsoft/MAI-DS-R1"}],"link_count":2,"sections":9},{"id":"non-transitivity-llm-judge-2025","title":"Investigating Non-Transitivity in LLM-as-a-Judge","year":2025,"venue":"ICML 2025","authors":["Yi Xu","Laura Ruis","Tim Rocktäschel","Robert Kirk"],"authors_zh":"Yi Xu, Laura Ruis, Tim Rocktäschel, Robert Kirk","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"LLM 评测器可靠性与排名协议审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Shows ranking instability when pairwise LLM judge preferences violate transitivity.","审计两两 LLM 判决的非传递性，并以锦标赛聚合降低排名对基线的敏感度。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://proceedings.mlr.press/v267/xu25w.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yix8/llm-nontransitivity"}],"link_count":3,"sections":9},{"id":"bon-coverage-optimality-2025","title":"Is Best-of-N the Best of Them? Coverage, Scaling, and Optimality in Inference-Time Alignment","year":2025,"venue":"ICML 2025 (PMLR 267)","authors":["Audrey Huang","Adam Block","Qinghua Liu","Nan Jiang","Akshay Krishnamurthy","Dylan J. Foster"],"authors_zh":"Audrey Huang、Adam Block、Qinghua Liu、Nan Jiang、Akshay Krishnamurthy、Dylan J Foster","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning","mathematics","science","instruction_following","inference_time_alignment"],"tags":["track5","best-of-n","inference-time-alignment","reward-hacking","coverage","inference-time-pessimism"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["The paper formalizes when Best-of-N overoptimizes imperfect reward models and proposes InferenceTimePessimism, a chi-squared-regularized rejection sampler evaluated across math, science, and instruction-following tasks.","该论文形式化刻画了 Best-of-N 在何时会对不完美的奖励模型过度优化，并提出以卡方正则化的拒绝采样器 InferenceTimePessimism，在数学、科学与指令遵循任务上评测。"],"why":"It separates rollout coverage, proxy reward, true evaluation, candidate budget, and rejection decisions, exposing the fields needed to audit test-time selection traces instead of treating a final benchmark score as data-quality evidence.","primary_link":"https://proceedings.mlr.press/v267/huang25c.html","links":[],"link_count":3,"sections":9},{"id":"itbench-2025","title":"ITBench: Evaluating AI Agents across Diverse Real-World IT Automation Tasks","year":2025,"venue":"ICML 2025","authors":["Saurabh Jha","Rohan R. Arora","Yuji Watanabe","Takumi Yanagawa","Yinfang Chen","Jackson Clark","Bhavya Bhavya","Mudit Verma","Harshit Kumar","Hirokuni Kitahara","Noah Zheutlin","Saki Takano","Divya Pathak","Felix George","Xinbo Wu","Bekir O Turkkan","Gerard Vanloo","Michael Nidd","Ting Dai","Oishik Chatterjee","Pranjal Gupta","Suranjana Samanta","Pooja Aggarwal","Rong Lee","Jae-Wook Ahn","Debanjana Kar","Amit Paradkar","Yu Deng","Pratibha Moogi","Prateeti Mohapatra","Naoki Abe","Chandrasekhar Narayanaswami","Tianyin Xu","Lav R. Varshney","Ruchi Mahindru","Anca Sailer","Laura Shwartz","Daby Sow","Nicholas C. M. Fuller","Ruchir Puri"],"authors_zh":"Saurabh Jha, Rohan R. Arora, Yuji Watanabe, Takumi Yanagawa, Yinfang Chen, Jackson Clark, Bhavya Bhavya, Mudit Verma, Harshit Kumar, Hirokuni Kitahara, Noah Zheutlin, Saki Takano, Divya Pathak, Felix George, Xinbo Wu, Bekir O Turkkan, Gerard Vanloo, Michael Nidd, Ting Dai, Oishik Chatterjee, Pranjal Gupta, Suranjana Samanta, Pooja Aggarwal, Rong Lee, Jae-Wook Ahn, Debanjana Kar, Amit Paradkar, Yu Deng, Pratibha Moogi, Prateeti Mohapatra, Naoki Abe, Chandrasekhar Narayanaswami, Tianyin Xu, Lav R. Varshney, Ruchi Mahindru, Anca Sailer, Laura Shwartz, Daby Sow, Nicholas C. M. Fuller, Ruchir Puri","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","infrastructure","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation","agent_training","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent-trajectories","environment-interaction","tool-use","site-reliability-engineering","compliance-and-security-operations","finops","kubernetes"],"tags":["agent-environment","state-action-trajectories","tool-use","environmental-verification","mixed-verifier","failure-retention","SRE","CISO","FinOps","Kubernetes","ICML-2025"],"status":"partial","priority":"必读","paper_type_zh":"真实IT自动化智能体基准、环境框架与轨迹数据发布","best_for_zh":"研究环境交互数据、工具智能体评测、agent training与轨迹审计的读者","confidence":"high","one_line":["ITBench packages live IT incidents and controls as tool-using agent episodes with explicit stop/success contracts, then extends them with versioned offline snapshots and SRE traces.","ITBench把真实IT运维问题组织为可部署环境中的工具调用episode，并发布静态场景与SRE轨迹；其价值在于明确stop与success契约，但跨版本映射、回放、污染、隐私和许可仍未解决。"],"why":"It exposes the full reasoning-data interface—scenario provenance, partial observations, executable actions, environment feedback, terminal predicates, and failed/incomplete episodes—while also showing why release version, replay, judge, and contamination metadata must travel with any trajectory corpus used for training or evaluation.","primary_link":"https://proceedings.mlr.press/v267/jha25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/itbench-hub/ITBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ibm-research/ITBench-Lite"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/ibm-research/ITBench-Trajectories"},{"key":"project","label":["Project","项目主页"],"url":"https://research.ibm.com/publications/itbench-evaluating-ai-agents-across-diverse-real-world-it-automation-tasks"}],"link_count":10,"sections":9},{"id":"iterative-deepening-sampling-2025","title":"Iterative Deepening Sampling as Efficient Test-Time Scaling","year":2025,"venue":"arXiv preprint","authors":["Weizhe Chen","Sven Koenig","Bistra Dilkina"],"authors_zh":"Weizhe Chen、Sven Koenig、Bistra Dilkina（机构：南加州大学、加州大学欧文分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","iterative-sampling","self-correction","reasoning-efficiency"],"status":"verified","priority":"可读","paper_type_zh":"无需训练的迭代测试时采样研究","best_for_zh":"探索自我纠错如何分配推理时计算的读者。","confidence":"medium","one_line":["Iterative Deepening Sampling uses progressively deeper self-correction samples to improve fixed-model reasoning at test time.","迭代加深采样以逐步加深的自我纠错样本改善固定模型的测试时推理。"],"why":"It studies how to obtain more useful reasoning samples from a fixed model instead of merely increasing independent rollouts.","primary_link":"https://arxiv.org/abs/2502.05449","links":[],"link_count":2,"sections":9},{"id":"j4r-reasoning-judge-eis-grpo-2025","title":"J4R: Learning to Judge with Equivalent Initial State Group Relative Policy Optimization","year":2025,"venue":"arXiv 2025","authors":["Austin Xu","Yilun Zhou","Xuan-Phi Nguyen","Caiming Xiong","Shafiq Joty"],"authors_zh":"Austin Xu, Yilun Zhou, Xuan-Phi Nguyen, Caiming Xiong, Shafiq Joty","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reasoning-evaluation","llm-as-a-judge","preference-learning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"推理判别器强化学习与成对偏好评测基准论文","best_for_zh":"需要训练或评估自动评审模型在多跳、数学、领域与常识推理中区分正确和错误答案的研究者。","confidence":"high","one_line":["J4R trains a position-robust reasoning judge and releases 1,483 outcome-oriented response pairs across eight source benchmarks.","1,483组跨八基准的正负推理答案，提供偏好标签，适合训练可泛化的推理结果判别器。"],"why":"It provides a concrete pairwise supervision contract for testing whether a judge recognizes reasoning correctness rather than response position.","primary_link":"https://arxiv.org/abs/2505.13346","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SalesforceAIResearch/ReasoningJudgeBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Salesforce/ReasoningJudgeBench"}],"link_count":4,"sections":9},{"id":"ji2s-joint-influence-selection-2025","title":"JI²S: Joint Influence-Aware Instruction Data Selection for Efficient Fine-Tuning","year":2025,"venue":"EMNLP 2025","authors":["Jingyu Wei","Bo Liu","Tianjiao Wan","Baoyun Peng","Xingkong Ma","Mengmeng Guo"],"authors_zh":"Jingyu Wei, Bo Liu, Tianjiao Wan, Baoyun Peng, Xingkong Ma, Mengmeng Guo","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["optimizer_scaffold"],"domains":["instruction-tuning","general"],"tags":["instruction-selection","joint-influence","lora","sft"],"status":"verified","priority":"可读","paper_type_zh":"联合影响感知的指令数据选择研究","best_for_zh":"研究紧凑指令微调中冗余感知选择的读者。","confidence":"high","one_line":["JI²S selects a compact instruction-tuning subset by combining marginal and pairwise joint influence relative to a human-preference proxy.","JI²S 相对人工偏好代理结合边际与成对联合影响，选择紧凑的指令微调子集。"],"why":"It makes interaction between records, not only individual quality, part of the SFT admission decision.","primary_link":"https://aclanthology.org/2025.emnlp-main.26/","links":[],"link_count":2,"sections":9},{"id":"judge-anything-any-modality-2025","title":"Judge Anything: MLLM as a Judge Across Any Modality","year":2025,"venue":"arXiv preprint","authors":["Shu Pu","Yaochen Wang","Dongping Chen","Yuhang Chen","Guohao Wang","Qi Qin","Zhongyi Zhang","Zhiyuan Zhang","Zetong Zhou","Shuang Gong","Yi Gui","Yao Wan","Philip S. Yu"],"authors_zh":"Shu Pu、Yaochen Wang、Dongping Chen、Yuhang Chen、Guohao Wang、Qi Qin、Zhongyi Zhang、Zhiyuan Zhang、Zetong Zhou、Shuang Gong、Yi Gui、Yao Wan、Philip S. Yu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["JudgeAnything benchmarks multimodal judges across any-to-any tasks with human judgments, rubrics, and pairwise or score-based protocols.","JudgeAnything 以人工判定、量规及成对或评分协议，评测多模态模型在任意模态任务上的评判能力。"],"why":"JudgeAnything benchmarks multimodal judges across any-to-any tasks with human judgments, rubrics, and pairwise or score-based protocols.","primary_link":"https://arxiv.org/abs/2503.17489","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/URRealHero/JudgeAnything"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/pudashi/JudgeAnything"},{"key":"project","label":["Project","项目主页"],"url":"https://urrealhero.github.io/judgeanythingweb/"}],"link_count":4,"sections":9},{"id":"judge-verdict-benchmark-2026","title":"Judge's Verdict: A Comprehensive Analysis of LLM Judge Capability Through Human Agreement","year":2025,"venue":"arXiv preprint; under review at ICLR 2026","authors":["Steve Han","Gilberto Titericz Junior","Tom Balough","Wenfei Zhou"],"authors_zh":"请以官方论文作者列表为准。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","llm-as-a-judge","human-agreement"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Public submission materials define a benchmark for factual judge reliability.","公开的投稿材料界定了一个衡量事实性评审可靠度的基准。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://arxiv.org/abs/2510.09738","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/nvidia/judges-verdict"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/judges-verdict"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/nvidia/judges-verdict"}],"link_count":5,"sections":9},{"id":"judgelm-2025","title":"JudgeLM: Fine-tuned Large Language Models are Scalable Judges","year":2025,"venue":"ICLR 2025","authors":["Lianghui Zhu","Xinggang Wang","Xinlong Wang"],"authors_zh":"Lianghui Zhu、Xinggang Wang、Xinlong Wang","tracks":["audit_failure_contamination_verifier_attacks","judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["audit","evaluation"],"construction_layer":["release_audit"],"domains":["contamination","verifier-audit","evaluation-reliability"],"tags":["audit","llm-as-a-judge","evaluation-reliability"],"status":"verified","priority":"可读","paper_type_zh":"LLM 裁判训练与可靠性审计论文","best_for_zh":"需要构建或审计开放式回答评估器、并控制位置与参考答案依赖偏差的研究者。","confidence":"high","one_line":["JudgeLM releases judge-training examples and a benchmark while explicitly measuring position, knowledge, and format bias.","JudgeLM 用 GPT-4 蒸馏的成对判决数据训练可本地部署的 LLM 裁判，并针对位置、知识与格式偏差设计干预。"],"why":"It turns LLM-as-a-judge failure modes into auditable data and evaluation objects rather than opaque prompt behavior.","primary_link":"https://openreview.net/forum?id=ZVb8VfDMbC","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/baaivision/judgelm"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/BAAI/JudgeLM-100K"}],"link_count":4,"sections":9},{"id":"position-bias-judges-2025","title":"Judging the Judges: A Systematic Study of Position Bias in LLM-as-a-Judge","year":2025,"venue":"IJCNLP-AACL 2025","authors":["Lin Shi","Chiyu Ma","Wenhua Liang","Xingjian Diao","Weicheng Ma","Soroush Vosoughi"],"authors_zh":"Lin Shi, Chiyu Ma, Wenhua Liang, Xingjian Diao, Weicheng Ma, Soroush Vosoughi","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"LLM judge 位置偏差系统审计论文","best_for_zh":"需要选择、审计或校准自动评测 judge 的研究者。","confidence":"medium","one_line":["Accepted systematic audit of position bias with public proceedings materials.","以重复查询与候选换序审计 LLM judge 的位置偏差、稳定性与偏好方向。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.ijcnlp-long.18/","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/2025.ijcnlp-long.18.pdf"}],"link_count":2,"sections":9},{"id":"judging-judges-gem-2025","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","year":2025,"venue":"GEM² 2025","authors":["Aman Singh Thakur","Kartik Choudhary","Venkat Srinik Ramayapally","Sankaran Vaidyanathan","Dieuwke Hupkes"],"authors_zh":"Aman Singh Thakur, Kartik Choudhary, Venkat Srinik Ramayapally, Sankaran Vaidyanathan, Dieuwke Hupkes","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["llm-as-a-judge","human-alignment","audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Measures how LLM judges agree with humans and how prompting and references destabilize their verdicts.","量化 LLM 评审与人工的一致性，并揭示提示词、参考答案顺序和宽松偏差造成的失稳。"],"why":"It demonstrates that a judge's ranking ability can conceal poorly calibrated human-aligned decisions.","primary_link":"https://aclanthology.org/2025.gem-1.33/","links":[],"link_count":2,"sections":9},{"id":"kernelbench-2025","title":"KernelBench: Can LLMs Write Efficient GPU Kernels?","year":2025,"venue":"arXiv preprint; official repository labels ICML 2025","authors":["Anne Ouyang","Simon Guo","Simran Arora","Alex L. Zhang","William Hu","Christopher Ré","Azalia Mirhoseini"],"authors_zh":"Anne Ouyang 等（Stanford University、Princeton University）","tracks":["benchmarks_evaluation_surfaces","programmatically_verifiable_outcome_data"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["gpu-programming","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","gpu-programming"],"status":"verified","priority":"可读","paper_type_zh":"ICML 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["KernelBench exposes CUDA kernel generation with correctness and speed checks as an auditable evaluation surface.","KernelBench 把带正确性与加速比检查的 CUDA 核函数生成做成可审计的评测面。"],"why":"Executable performance-sensitive reasoning surface.","primary_link":"https://arxiv.org/abs/2502.10517","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ScalingIntelligence/KernelBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScalingIntelligence/KernelBench"},{"key":"project","label":["Project","项目主页"],"url":"https://scalingintelligence.stanford.edu/blogs/kernelbench/"}],"link_count":7,"sections":9},{"id":"kimi-k1-5-2025","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","year":2025,"venue":"arXiv preprint","authors":["Kimi Team"],"authors_zh":"Kimi Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["mathematics","coding","general_reasoning","vision_language","long_context"],"tags":["kimi-k1-5","frontier-report","data-disclosure-ledger","long-cot","rlvr","reward-model","code-test-generation","long2short"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态强化学习技术报告与数据披露台账","best_for_zh":"需要审计前沿推理模型的 RL prompt、奖励/验证器、Long-CoT 与 long2short 配方、代码环境及缺失发布工件的读者","confidence":"medium","one_line":["Kimi k1.5 discloses a multimodal RL pipeline with prompt filtering, verified long-CoT warmup, math reward models, synthesized code tests, long2short preference/RL methods, and partial-rollout infrastructure, but releases no model, training data, reward data, or executable training stack.","Kimi k1.5 披露了包含 prompt 过滤、已验证 Long-CoT warmup、数学奖励模型、合成代码测试、long2short 偏好/RL 方法与 partial-rollout 基础设施的多模态 RL 管线，但未发布模型、训练数据、奖励数据或可执行训练栈。"],"why":"It lets Track 12 readers audit a frontier report at the level of data and feedback interfaces while making clear that detailed methods do not replace source manifests, released validators, licenses, splits, decontamination, or reproducible artifacts.","primary_link":"https://arxiv.org/abs/2501.12599","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/MoonshotAI/Kimi-k1.5"}],"link_count":4,"sections":9},{"id":"kimi-k2-open-agentic-intelligence-2025","title":"Kimi K2: Open Agentic Intelligence","year":2025,"venue":"arXiv preprint","authors":["Kimi Team"],"authors_zh":"Kimi Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward","pairwise_preference"],"training_use":["sft","rlvr","agent_training","preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["agentic_tool_use","software_engineering","coding","mathematics","reasoning","safety","general"],"tags":["kimi-k2","frontier-report","data-disclosure-ledger","mcp","agentic-trajectories","execution-sandbox","rlvr","self-critique"],"status":"partial","priority":"必读","paper_type_zh":"前沿智能体模型技术报告与数据披露台账","best_for_zh":"需要审计 MCP 工具、智能体轨迹、真实执行环境、RLVR 与自评奖励披露边界的读者","confidence":"medium","one_line":["Kimi K2 reports 3,000+ real MCP and 20,000+ synthetic tools feeding rubric-filtered trajectories, execution environments, RLVR, and self-critique without releasing data, verifier implementation, or audit artifacts.","Kimi K2 报告称 3,000+ 个真实 MCP 工具和 20,000+ 个合成工具经由 rubric 过滤轨迹、执行环境、RLVR 与 self-critique 反馈进入后训练，但未公开数据、verifier 实现或审计工件。"],"why":"Gives unusually concrete interfaces for agent-data and feedback construction while separating public weights from source provenance, calibration, rights, and reproducibility.","primary_link":"https://arxiv.org/abs/2507.20534","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MoonshotAI/Kimi-K2"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/moonshotai/Kimi-K2-Instruct"},{"key":"project","label":["Project","项目主页"],"url":"https://moonshotai.github.io/Kimi-K2/"}],"link_count":5,"sections":9},{"id":"kimi-dev-2025","title":"Kimi-Dev: Agentless Training as Skill Prior for SWE-Agents","year":2025,"venue":"arXiv preprint","authors":["Zonghan Yang","Shengjie Wang","Kelin Fu","Wenyang He","Weimin Xiong","Yibo Liu","Yibo Miao","Bofei Gao","Yejie Wang","Yingwei Ma","Yanhao Li","Yue Liu","Zhenxing Hu","Kaitai Zhang","Shuyi Wang","Huarong Chen","Flood Sung","Yang Liu","Yang Gao","Zhilin Yang","Tianyu Liu"],"authors_zh":"Zonghan Yang 等（Moonshot AI）","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","data_release","verifier_reward","agent_environment","construction_recipe","scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","agent_training","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["software_engineering","repository_issue_resolution","bug_fixing","test_generation","code_editing","repository_navigation","agentic_coding","python"],"tags":["kimi-dev","moonshot-ai","software-engineering","repository-editing","agentless","skill-prior","swe-agent","github-pull-requests","synthetic-trajectories","docker","binary-reward","rlvr","bugfixer","testwriter","test-time-self-play","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"软件工程智能体数据构造与可验证强化学习技术报告","best_for_zh":"研究 repository issue-to-patch 数据、Docker verifier、Agentless 技能先验到 agent 的迁移，以及数据披露审计的读者","confidence":"high","one_line":["Kimi-Dev turns GitHub patch and simulated file-tool data into an Agentless skill prior, then applies full-test-suite binary RLVR and a reported 5,016-trajectory agent bridge, but withholds the training corpus, Docker provenance, exact splits, failures, and adapted checkpoint.","Kimi-Dev 将 GitHub 补丁与模拟文件工具数据训练成 Agentless 技能先验，再用完整任务测试的二元 RLVR 和论文所述 5,016 条 agent 轨迹连接 SWE-Agent；但训练语料、Docker 谱系、精确划分、失败 rollout 与适配后 checkpoint 未发布。"],"why":"It separates patch skill acquisition from environment-grounded RL and agent adaptation, making the paper useful both as a construction recipe and as an audit case for executable software-engineering data.","primary_link":"https://arxiv.org/abs/2509.23045","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MoonshotAI/Kimi-Dev"},{"key":"data","label":["Data","数据"],"url":"https://github.com/MoonshotAI/Kimi-Dev/tree/master/resources"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/moonshotai/Kimi-Dev-72B"},{"key":"project","label":["Project","项目主页"],"url":"https://moonshotai.github.io/Kimi-Dev/"}],"link_count":6,"sections":9},{"id":"kimi-vl-technical-report-2025","title":"Kimi-VL Technical Report","year":2025,"venue":"arXiv preprint","authors":["Kimi Team"],"authors_zh":"Kimi Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["multimodal_reasoning","computer_vision","agent","mathematics"],"tags":["kimi-vl","frontier-report","data-disclosure-ledger","multimodal","long-cot","reinforcement-learning","reward-model","agent-trajectory"],"status":"partial","priority":"可读","paper_type_zh":"前沿多模态模型技术报告与数据披露账本","best_for_zh":"需要审计前沿多模态模型的数据来源、长 CoT、奖励反馈、agent 轨迹及发布边界的读者","confidence":"medium","one_line":["Kimi-VL reports a mixed multimodal pipeline from human and model-generated SFT to Kimi k1.5 long-CoT filtering and ground-truth-centered RL, while releasing weights and inference material but not the training data, reward artifacts, agent environment, or audit records.","Kimi-VL 报告了从人工和模型生成 SFT、Kimi k1.5 长 CoT 筛选到以真值为中心的 RL 的混合多模态管线；其公开了权重和推理材料，但未公开训练数据、奖励工件、agent 环境或审计记录。"],"why":"It separates a detailed frontier-model narrative from the missing source manifests, validator specifications, licenses, splits, decontamination evidence, and replayable environments that Track 12 readers need to assess reuse and auditability.","primary_link":"https://arxiv.org/abs/2504.07491","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MoonshotAI/Kimi-VL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/moonshotai/Kimi-VL-A3B-Thinking-2506"}],"link_count":5,"sections":9},{"id":"kimina-prover-preview-2025","title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["Haiming Wang","Mert Unsal","Xiaohan Lin","Mantas Baksys","Junqi Liu","Marco Dos Santos","Flood Sung","Marina Vinyes","Zhenzhe Ying","Zekai Zhu","Jianqiao Lu","Hugues de Saxce","Bolton Bailey","Chendong Song","Chenjun Xiao","Dehao Zhang","Ebony Zhang","Frederick Pu","Han Zhu","Jiawei Liu","Jonas Bayer","Julien Michel","Longhui Yu","Leo Dreyfus-Schmidt","Lewis Tunstall","Luigi Pagani","Moreira Machado","Pauline Bourigault","Ran Wang","Stanislas Polu","Thibaut Barroyer","Wen-Ding Li","Yazhe Niu","Yann Fleureau","Yangyang Hu","Zhouliang Yu","Zihan Wang","Zhilin Yang","Zhengying Liu","Jia Li"],"authors_zh":"Haiming Wang 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","distillation","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","formal_theorem_proving","lean4","autoformalization"],"tags":["kimina-prover","formal-reasoning","lean4","terminal-verification","rlvr","autoformalization","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"前沿形式化推理技术报告与数据披露账本","best_for_zh":"需要区分 Lean 整证明终态验证、自动形式化语义审查与交互式步骤反馈，并审计 RL 复现边界的读者","confidence":"high","one_line":["Kimina-Prover discloses a 200K Lean prompt construction, about 20K cold starts, whole-proof compiler-terminal RL, and public distills/tooling, but not core prompts, rollouts, rewards, environment pins, or full lineage.","Kimina-Prover 披露了 20 万条 Lean prompt 构建、约 2 万条冷启动轨迹、整证明的 Lean 编译终态 RL，以及公开的蒸馏模型/工具；但未公开核心 prompt、rollout、奖励、环境版本锁定或完整谱系。"],"why":"It distinguishes whole-proof terminal programmatic feedback from imperfect LLM/human-reviewed autoformalization and from interactive step feedback, showing why released artifacts do not reproduce the full pipeline.","primary_link":"https://arxiv.org/abs/2504.11354","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MoonshotAI/Kimina-Prover-Preview"},{"key":"data","label":["Data","数据"],"url":"https://github.com/MoonshotAI/Kimina-Prover-Preview/blob/master/minif2f_test_solved_filtered_0710.zip"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/AI-MO/Kimina-Prover-Preview-Distill-7B"}],"link_count":6,"sections":9},{"id":"klear-reasoner-advancing-reasoning-capability-via-gradient-preserving-clipping-policy-op","title":"Klear-Reasoner: Advancing Reasoning Capability via Gradient-Preserving Clipping Policy Optimization","year":2025,"venue":"arXiv","authors":["Zhenpeng Su","Leiyu Pan","Xue Bai","Dening Liu","Guanting Dong","Jiaming Huang","Minxuan Lv","Wenping Hu","Fuzheng Zhang","Kun Gai","Guorui Zhou"],"authors_zh":"Zhenpeng Su, Leiyu Pan, Xue Bai, Dening Liu, Guanting Dong, Jiaming Huang, Minxuan Lv, Wenping Hu, Fuzheng Zhang, Kun Gai, Guorui Zhou","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** Klear-Reasoner combines difficult MathSub-30K prompts with GPPO to use failed RLVR trajectories more effectively.","Klear-Reasoner 用 MathSub-30K 困难题和 GPPO 共同提高 RLVR 对负轨迹的利用。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2508.07629","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/suu990901/KlearReasoner"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Kwai-Klear/KlearReasoner-MathSub-30K"}],"link_count":3,"sections":9},{"id":"llm-abstention-survey-2025","title":"Know Your Limits: A Survey of Abstention in Large Language Models","year":2025,"venue":"TACL 2025","authors":["Bingbing Wen","Jihan Yao","Shangbin Feng","Chenjun Xu","Yulia Tsvetkov","Bill Howe","Lucy Lu Wang"],"authors_zh":"Bingbing Wen 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["abstention","hallucination","safety"],"tags":["foundations-and-primers","abstention","tacl-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"拒答能力综述","best_for_zh":"关注幻觉缓解、安全回答和选择性预测的读者。","confidence":"high","one_line":["A TACL 2025 survey organizing LLM abstention through the query, model, and human values.","从查询、模型和人类价值三个角度梳理大语言模型何时应当拒答。"],"why":"It treats declining to answer as a decision that must be both useful and accountable to context.","primary_link":"https://aclanthology.org/2025.tacl-1.26/","links":[],"link_count":2,"sections":9},{"id":"kodcode-2025","title":"KodCode: A Diverse, Challenging, and Verifiable Synthetic Dataset for Coding","year":2025,"venue":"Findings of ACL 2025","authors":["Zhangchen Xu","Yang Liu","Yueqin Yin","Mingyuan Zhou","Radha Poovendran"],"authors_zh":"Zhangchen Xu、Yang Liu、Yueqin Yin、Mingyuan Zhou、Radha Poovendran","tracks":["instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["code-generation","algorithms","data-structures","package-use","competitive-programming"],"tags":["instruction-demonstration-rationale","synthetic-code-data","unit-test-verification","deepseek-r1-distillation","arxiv-2503.02951","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"可执行校验的合成编程数据集与推理 SFT 配方","best_for_zh":"适合构造单元测试验收的编程 SFT/RL 数据，或审计同源生成的答案与测试。","confidence":"high","one_line":["KodCode releases large CC BY-NC coding triplets and 268,211 verified R1-SFT records with tests, trial statistics, difficulty, and contamination fields.","KodCode 公开大规模 CC BY-NC 编程三元组与 268,211 条经单元测试验收的 R1-SFT 记录，并保留试验、难度和污染字段。"],"why":"It makes both synthetic-code supervision and its executable acceptance evidence inspectable across twelve source families and several difficulty bands.","primary_link":"https://aclanthology.org/2025.findings-acl.365/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/KodCode-AI/kodcode"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/KodCode/KodCode-V1-SFT-R1"},{"key":"project","label":["Project","项目主页"],"url":"https://kodcode-ai.github.io/"}],"link_count":7,"sections":9},{"id":"kvasir-vqa-x1-2025","title":"Kvasir-VQA-x1: A Multimodal Dataset for Medical Reasoning and Robust MedVQA in Gastrointestinal Endoscopy","year":2025,"venue":"Data Engineering in Medical Imaging, Springer","authors":["Sushant Gautam","Michael A. Riegler","Pål Halvorsen"],"authors_zh":"Sushant Gautam、Michael A. Riegler、Pål Halvorsen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","benchmark"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["gastrointestinal-endoscopy","medical-visual-question-answering","multimodal-clinical-reasoning"],"tags":["medical-vqa-data","clinical-question-merging","multimodal-instruction-data","robustness-augmentation","arxiv-2506.09958","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放医学多模态指令数据集与基准研究","best_for_zh":"临床视觉问答微调、课程学习实验与视觉扰动鲁棒性审计","confidence":"high","one_line":["Kvasir-VQA-x1 turns atomic GI-endoscopy QAs into 159,549 expert-validated, complexity-stratified multimodal demonstrations for clinical VQA training and robustness evaluation.","Kvasir-VQA-x1 将胃肠内镜原子问答扩展为 159,549 条经专家验证、按复杂度分层的多模态示范，用于临床视觉问答训练和鲁棒性评估。"],"why":"Existing gastrointestinal VQA data emphasizes atomic recognition and offers limited compositional reasoning or controlled robustness testing.","primary_link":"https://arxiv.org/abs/2506.09958","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Simula/Kvasir-VQA-x1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SimulaMet/Kvasir-VQA-x1"}],"link_count":4,"sections":9},{"id":"l1-controlling-thinking-length-2025","title":"L1: Controlling How Long A Reasoning Model Thinks With Reinforcement Learning","year":2025,"venue":"COLM 2025","authors":["Pranjal Aggarwal","Sean Welleck"],"authors_zh":"unknown","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","test_time_compute","evaluation"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematical_reasoning","logical_reasoning","general_knowledge"],"tags":["length-control","lcpo","grpo","rlvr","test-time-compute","budget-conditioning","reasoning-traces","exact-length","maximum-length"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"high","one_line":["L1 trains DeepScaleR-based policies with programmatic final-answer and token-budget rewards so a prompt can request either an exact reasoning length or a maximum token ceiling.","L1 以正确性和长度约束奖励使推理长度成为提示词可控的测试时计算预算。"],"why":"It makes inference budget a first-class field of the rollout and reward contract: each source problem is paired with a requested length, the policy generates an on-policy trace, and correctness plus realized length determine reward. For Track 5, this separates learned budget-conditioned trajectories from hard truncation, while the missing complete rollout/reward ledger limits data-level reuse and audit.","primary_link":"https://openreview.net/forum?id=4jdIxXBNve","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/cmu-l3/l1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/l3lab/l1"},{"key":"project","label":["Project","项目主页"],"url":"https://cmu-l3.github.io/l1/"}],"link_count":7,"sections":9},{"id":"causal-inference-llm-survey-2025","title":"Large Language Models and Causal Inference in Collaboration: A Comprehensive Survey","year":2025,"venue":"Findings of NAACL 2025","authors":["Xiaoyu Liu","Paiheng Xu","Junda Wu","Jiaxin Yuan","Yifan Yang","Yuhang Zhou","Fuxiao Liu","Tianrui Guan","Haoliang Wang","Tong Yu","Julian McAuley","Wei Ai","Furong Huang"],"authors_zh":"Xiaoyu Liu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["causal-inference","reasoning","fairness","explainability"],"tags":["foundations-and-primers","causal-inference","naacl-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"因果推断与大语言模型综述","best_for_zh":"关注因果推理、公平性、可解释性或因果发现的读者。","confidence":"high","one_line":["A Findings NAACL 2025 survey of two-way collaboration between causal inference and language models.","综述因果推断与大语言模型双向协作的研究。"],"why":"It distinguishes using causal ideas to evaluate language models from using language models to assist causal tasks.","primary_link":"https://aclanthology.org/2025.findings-naacl.427/","links":[],"link_count":2,"sections":9},{"id":"alignment-potential-preference-selection-2025","title":"Larger or Smaller Reward Margins to Select Preferences for LLM Alignment?","year":2025,"venue":"ICML 2025","authors":["Kexin Huang","Junkang Wu","Ziqian Chen","Xue Wang","Jinyang Gao","Bolin Ding","Jiancan Wu","Xiangnan He","Xiang Wang"],"authors_zh":"Kexin Huang、Junkang Wu、Ziqian Chen 等（中国科学技术大学、独立研究者）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-following"],"tags":["data-selection","preference-learning","reward-margin","simpo","alignment"],"status":"verified","priority":"必读","paper_type_zh":"偏好间隔驱动的数据选择与对齐研究","best_for_zh":"适合需要从固定偏好对中选择更有训练价值记录的读者。","confidence":"high","one_line":["Alignment Potential selects preference pairs whose current policy margin differs most from the reward-model target margin.","对齐潜力指标选择当前策略间隔与奖励目标间隔差距最大的偏好对。"],"why":"It changes data use from selecting only easy or only high-reward pairs to selecting pairs with the most estimated room for alignment.","primary_link":"https://proceedings.mlr.press/v267/huang25al.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Hesse73/Alignment-Potential-Metric"}],"link_count":3,"sections":9},{"id":"lastingbench-defend-benchmarks-against-knowledge-leakage-2025","title":"LastingBench: Defend Benchmarks Against Knowledge Leakage","year":2025,"venue":"Findings of EMNLP 2025","authors":["Yixiong Fang","Tianran Sun","Yuling Shi","Min Wang","Xiaodong Gu"],"authors_zh":"Yixiong Fang、Tianran Sun、Yuling Shi、Min Wang、Xiaodong Gu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🧱 Foundations: Instruction, Preference, and Alignment Data map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","聚焦如何抵抗知识泄漏并延长基准有效期，是训练—测试重叠审计的重要工具。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://aclanthology.org/2025.findings-emnlp.993/","links":[],"link_count":2,"sections":9},{"id":"lazyreview-peer-review-lazy-thinking-2025","title":"LazyReview: A Dataset for Uncovering Lazy Thinking in NLP Peer Reviews","year":2025,"venue":"ACL 2025","authors":["Sukannya Purkayastha","Zhuang Li","Anne Lauscher","Lizhen Qu","Iryna Gurevych"],"authors_zh":"Sukannya Purkayastha、Zhuang Li、Anne Lauscher、Lizhen Qu、Iryna Gurevych","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"可读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["LazyReview uses expert and silver review segments to identify shallow reasoning in NLP peer reviews.","用专家与银标评审片段识别 NLP 同行评审中缺少实质论证的惰性思考。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://aclanthology.org/2025.acl-long.165/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UKPLab/arxiv2025-lazy-review"},{"key":"data","label":["Data","数据"],"url":"https://tudatalib.ulb.tu-darmstadt.de/items/ce21aeab-85bc-4d26-b9cc-322f34c6f64d"},{"key":"project","label":["Project","项目主页"],"url":"https://www.ukp.tu-darmstadt.de/data/lazyreview/"}],"link_count":5,"sections":9},{"id":"leaky-thoughts-2025","title":"Leaky Thoughts","year":2025,"venue":"EMNLP 2025","authors":[],"authors_zh":"Tommaso Green, Martin Gubri, Haritz Puerto, Sangdoo Yun, Seong Joon Oh","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure","benchmark"],"verification_contract":["judgment_required","environmental"],"supervision_granularity":["step_level","full_episode"],"training_use":["evaluation","safety_alignment","audit"],"construction_layer":["release_audit"],"domains":["privacy","agent","security"],"tags":[],"status":"verified","priority":"必读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Leaky Thoughts shows that reasoning traces from personal-agent settings can expose sensitive user data through prompt injection or accidental leakage.","显示个人智能体场景中的推理痕迹会被提示注入或意外输出泄露，把 CoT 直接转化为隐私攻击面。"],"why":"It turns chain-of-thought and test-time compute into a privacy audit problem: more internal reasoning can increase utility while enlarging the attack surface.","primary_link":"https://aclanthology.org/2025.emnlp-main.1347/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/parameterlab/leaky_thoughts"}],"link_count":3,"sections":9},{"id":"learn-by-interact-2025","title":"Learn-by-interact: A Data-Centric Framework For Self-Adaptive Agents in Realistic Environments","year":2025,"venue":"ICLR 2025","authors":["Hongjin Su","Ruoxi Sun","Jinsung Yoon","Pengcheng Yin","Tao Yu","Sercan Ö. Arık"],"authors_zh":"Hongjin Su、Ruoxi Sun、Jinsung Yoon、Pengcheng Yin、Tao Yu、Sercan Ö. Arık","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","model_report","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","search_substrate"],"domains":["software_engineering_agents","web_navigation_agents","desktop_computer_use","data_science_engineering_agents"],"tags":["learn-by-interact","agent-trajectories","backward-construction","environment-interaction","synthetic-data","llm-committee","agentic-retrieval","supervised-fine-tuning"],"status":"partial","priority":"必读","paper_type_zh":"智能体交互数据构造方法与扩展研究","best_for_zh":"研究如何把不完美的环境交互转化为 SFT 或智能体训练数据，以及审计目标重标注、模型委员会筛选、相关子轨迹和未开放工件带来的风险","confidence":"high","one_line":["The recipe expands 85,905 documentation-seeded agent episodes into 1,456,808 instruction-subtrajectory candidates and reports 440,008 committee-filtered examples, but releases neither the corpus nor its implementation.","Learn-by-interact 将 85,905 条由文档任务驱动的智能体交互轨迹扩展为 1,456,808 个指令—子轨迹候选，经双模型一致筛选后报告 440,008 条样本，但未发布语料、实现或模型。"],"why":"It exposes a reusable pattern for turning imperfect environment exploration into post-training data, while making clear that instruction-behavior alignment, original-task success, provenance, and open release are separate audit questions.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/file/dd745a0c4f91fe91866fe6788be9cc28-Paper-Conference.pdf","links":[],"link_count":6,"sections":9},{"id":"learning-how-hard-to-think-2025","title":"Learning How Hard to Think: Input-Adaptive Allocation of LM Computation","year":2025,"venue":"ICLR 2025","authors":["Mehul Damani","Idan Shenfeld","Andi Peng","Andreea Bobu","Jacob Andreas"],"authors_zh":"Mehul Damani、Idan Shenfeld、Andi Peng、Andreea Bobu、Jacob Andreas（机构：麻省理工学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","code-generation","general-reasoning"],"tags":["test-time-compute","adaptive-allocation","best-of-k","routing","marginal-reward"],"status":"verified","priority":"必读","paper_type_zh":"输入自适应测试时计算分配研究","best_for_zh":"构建共享推理服务、且并非每个请求都应获得相同样本数或昂贵解码路径的读者。","confidence":"high","one_line":["This work predicts each prompt’s marginal return from extra decoding and uses those predictions to allocate sampling or strong-model calls under a shared compute budget.","该工作预测每个提示从额外解码中获得的边际收益，并在共享计算预算下分配采样次数或强模型调用。"],"why":"It gives a general optimization formulation for deciding which inputs deserve more test-time compute, rather than treating difficulty as a post-hoc statistic.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/ff414825df833edb8b1839e3d5d495e9-Abstract-Conference.html","links":[],"link_count":3,"sections":9},{"id":"guided-rest-2025","title":"Learning to Better Search with Language Models via Guided Reinforced Self-Training","year":2025,"venue":"NeurIPS 2025","authors":["Seungyong Moon","Bumsoo Park","Hyun Oh Song"],"authors_zh":"Seungyong Moon, Bumsoo Park, Hyun Oh Song","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["general_reasoning","post-training"],"tags":["reinforced-self-training","subgoals","reference-code","reward-filtering"],"status":"partial","priority":"可读","paper_type_zh":"特权引导的搜索轨迹构造与自训练方法","best_for_zh":"适合研究搜索轨迹蒸馏、迭代自训练、代码修复反馈和特权信息泄漏审计的读者","confidence":"medium","one_line":["Guided-ReST uses optimal solutions as training-time landmarks to repair, verify, and distill current-policy search traces.","Guided-ReST 用最优解作为仅构造期可见的局部地标，修补并筛选策略自身的搜索轨迹；其价值在于展示特权引导蒸馏配方，但未发布冻结轨迹语料。"],"why":"It makes the privileged construction view, the filtered or masked training view, and the unassisted inference view explicit, which is central to auditing search-generated reasoning data.","primary_link":"https://arxiv.org/abs/2410.02992","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/snu-mllab/guided-rest"}],"link_count":4,"sections":9},{"id":"sol-ver-2025","title":"Learning to Solve and Verify: A Self-Play Framework for Code and Test Generation","year":2025,"venue":"NeurIPS 2025 Workshop on Deep Learning for Code in the Agentic Era","authors":["Zi Lin","Sheng Shen","Ilia Kulikov","Jingbo Shang","Jason Weston","Yixin Nie"],"authors_zh":"Zi Lin、Sheng Shen、Ilia Kulikov、Jingbo Shang、Jason Weston、Yixin Nie","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["code_generation","test_generation"],"tags":["solver-verifier","code-self-play","unit-test-generation","execution-filtering","synthetic-preferences","iterative-sft-dpo"],"status":"partial","priority":"可读","paper_type_zh":"代码与测试联合自博弈数据构建","best_for_zh":"研究可执行反馈、代码自博弈、测试生成和合成偏好数据的读者","confidence":"high","one_line":["Sol-Ver uses one Llama 3.1 8B model to co-generate Python solutions and unit tests, turning execution agreement into Solver/Verifier SFT records and DPO pairs over three self-play rounds.","Sol-Ver 让同一个 Llama 3.1 8B 同时生成代码与单元测试，以执行一致性构造 SFT 正例和 DPO 胜负对；其价值在闭环配方，而非未发布语料的直接复用。"],"why":"It makes executable feedback a closed-loop data-construction contract while exposing how weak generated tests, correlated roles, and unreleased lineage can turn selection errors into training data.","primary_link":"https://arxiv.org/abs/2502.14948","links":[],"link_count":5,"sections":9},{"id":"lemaj-legal-judge-2025","title":"LeMAJ (Legal LLM-as-a-Judge): Bridging Legal Reasoning and LLM Evaluation","year":2025,"venue":"NLLP 2025","authors":["Joseph Enguehard","Morgane Van Ermengem","Kate Atkinson","Sujeong Cha","Arijit Ghosh Chowdhury","Prashanth Kallur Ramaswamy","Jeremy Roghair","Hannah R Marlowe","Carina Suzana Negreanu","Kitty Boxall","Diana Mincu"],"authors_zh":"Joseph Enguehard, Morgane Van Ermengem, Kate Atkinson, Sujeong Cha, Arijit Ghosh Chowdhury, Prashanth Kallur Ramaswamy, Jeremy Roghair, Hannah R Marlowe, Carina Suzana Negreanu, Kitty Boxall, Diana Mincu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["legal-llm-judge","legal-reasoning","audit"],"status":"verified","priority":"可读","paper_type_zh":"评测器、验证器、奖励或基准可靠性研究","best_for_zh":"需要审计法律回答自动评审可靠性的研究者。","confidence":"medium","one_line":["Decomposes legal answers into claims for reference-free judging validated against legal experts.","将法律回答拆为主张级数据点，以无参考评审同法律专家判断对齐。"],"why":"It makes the judgment unit and its legal-review assumptions inspectable.","primary_link":"https://aclanthology.org/2025.nllp-1.23/","links":[],"link_count":2,"sections":9},{"id":"lemma-learning-errors-mathematical-advancement-2025","title":"LEMMA: Learning from Errors for MatheMatical Advancement in LLMs","year":2025,"venue":"Findings of ACL 2025","authors":["Zhuoshi Pan","Yu Li","Honglin Lin","Qizhi Pei","Zinan Tang","Wei Wu","Chenlin Ming","H. Vicky Zhao","Conghui He","Lijun Wu"],"authors_zh":"Zhuoshi Pan、Yu Li、Honglin Lin、Qizhi Pei、Zinan Tang、Wei Wu、Chenlin Ming、H. Vicky Zhao、Conghui He、Lijun Wu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","self-correction","error-supervision"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要构造错误—修正过程监督、训练数学推理自纠错能力的研究者。","confidence":"high","one_line":["LEMMA turns diverse model mistakes into error-to-correction training pairs, teaching mathematical LLMs to repair their own reasoning without an external critique model.","LEMMA 将多样化模型错误组织为错误到正确解的反思连接，使数学模型无需外部批评器也能在生成中自主纠错。"],"why":"It preserves the first error and its repair path, making negative reasoning evidence directly usable for training rather than discarding it.","primary_link":"https://aclanthology.org/2025.findings-acl.605/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pzs19/LEMMA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/panzs19/LEMMA"}],"link_count":5,"sections":9},{"id":"bees-preference-data-selection-2025","title":"Less is More: Improving LLM Alignment via Preference Data Selection","year":2025,"venue":"NeurIPS 2025","authors":["Xun Deng","Han Zhong","Rui Ai","Fuli Feng","Zheng Wang","Xiangnan He"],"authors_zh":"Xun Deng、Han Zhong、Rui Ai、Fuli Feng、Zheng Wang、Xiangnan He（中国科学技术大学、北京大学、麻省理工学院、阿里云）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","summarization","dialogue"],"tags":["preference-data-selection","direct-preference-optimization","reward-margin","data-curation"],"status":"verified","priority":"必读","paper_type_zh":"偏好数据选择与直接偏好优化研究","best_for_zh":"适合为数据高效对齐训练清洗含噪偏好样本对的读者。","confidence":"high","one_line":["BeeS selects DPO pairs with high confidence across external and implicit reward margins, improving alignment while training on much less preference data.","BeeS 选择在外部与隐式奖励间隔上均具高置信度的 DPO 样本对，以更少偏好数据改善对齐。"],"why":"It demonstrates that preference-pair selection can improve both alignment quality and training efficiency without changing the downstream loss.","primary_link":"https://arxiv.org/abs/2502.14560","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xiangtanshi/DPO-Data-Selection"}],"link_count":4,"sections":9},{"id":"lessleak-bench-2025","title":"LessLeak-Bench: A First Investigation of Data Leakage in LLMs Across 83 Software Engineering Benchmarks","year":2025,"venue":"arXiv","authors":["Xin Zhou","Martin Weyssow","Ratnadira Widyasari","Ting Zhang","Junda He","Yunbo Lyu","Jianming Chang","Beiqi Zhang","Dan Huang","David Lo"],"authors_zh":"Xin Zhou, Martin Weyssow, Ratnadira Widyasari, Ting Zhang, Junda He, Yunbo Lyu, Jianming Chang, Beiqi Zhang, Dan Huang, David Lo","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["benchmark-contamination","data-leakage","audit"],"status":"verified","priority":"可读","paper_type_zh":"基准污染与数据泄漏审计研究","best_for_zh":"需要审计代码基准污染和评测偏差的研究者。","confidence":"high","one_line":["Measures and removes verified pre-training leakage across 83 software-engineering benchmarks.","测量并删除 83 个软件工程基准中经核验的预训练泄漏样本。"],"why":"It gives concrete evidence that benchmark contamination can inflate LLM evaluation and a release to mitigate it.","primary_link":"https://arxiv.org/abs/2502.06215","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/26PaperSubmission/lessleak-bench"}],"link_count":2,"sections":9},{"id":"self-braking-tuning-2025","title":"Let LRMs Break Free from Overthinking via Self-Braking Tuning","year":2025,"venue":"NeurIPS 2025","authors":["Haoran Zhao","Yuchen Yan","Yongliang Shen","Haolei Xu","Wenqi Zhang","Kaitao Song","Jian Shao","Weiming Lu","Jun Xiao","Yueting Zhuang"],"authors_zh":"Haoran Zhao、Yuchen Yan、Yongliang Shen、Haolei Xu、Wenqi Zhang、Kaitao Song、Jian Shao、Weiming Lu、Jun Xiao、Yueting Zhuang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","process_reward"],"training_use":["sft","process_supervision","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics"],"tags":["self-braking","overthinking","trace-truncation","sft"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹重写与自制动微调配方","best_for_zh":"需要审计长 CoT 截断、停止监督与 OpenR1-Math 派生数据边界的读者","confidence":"high","one_line":["SBT derives masked/truncated math traces with braking prompts from OpenR1-Math.","SBT 从 OpenR1-Math 推导截断或掩码的推理轨迹并加入 braking prompt，以监督模型在冗余推理前自行终止；派生样本和决策账本尚未确认独立发布。"],"why":"It exposes construction of accepted and masked trace segments without publishing the full ledger.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/hash/02d028c3bf5d4875b5a4e4b4311087f7-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ZJU-REAL/Self-Braking-Tuning"},{"key":"project","label":["Project","项目主页"],"url":"https://zju-real.github.io/SBT/"}],"link_count":6,"sections":9},{"id":"aops-instruct-liveaopsbench-2025","title":"Leveraging Online Olympiad-Level Math Problems for LLMs Training and Contamination-Resistant Evaluation","year":2025,"venue":"ICML 2025","authors":["Sadegh Mahdavi","Muchen Li","Kaiwen Liu","Christos Thrampoulidis","Leonid Sigal","Renjie Liao"],"authors_zh":"Sadegh Mahdavi、Muchen Li、Kaiwen Liu、Christos Thrampoulidis、Leonid Sigal、Renjie Liao","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["mathematics","olympiad_mathematics"],"tags":["aops-instruct","liveaopsbench","olympiad-math","forum-data","timestamp-split","contamination","solution-rewriting"],"status":"partial","priority":"必读","paper_type_zh":"论坛数学 SFT 数据构造与时间评测配方","best_for_zh":"研究数学 SFT、动态基准、去污染、数据沿袭和社区内容发布审计的读者","confidence":"medium","one_line":["AoPS-Instruct turns pre-2024 forum questions and community answers into 647,255 rewritten SFT pairs, while LiveAoPSBench applies stricter answer checks to newer timestamped posts.","AoPS-Instruct 将 2024 年前的论坛问题与社区解答转为 647,255 条重写 SFT 记录，LiveAoPSBench 则用更严格的答案检查筛选较新的时间戳帖子；训练快照、逐记录来源与内容复用权利仍不完整。"],"why":"It exposes a scalable forum-distillation and temporal-evaluation recipe while showing why model-ready records still need immutable snapshots, source lineage, verifier audits, contributor attribution, and documented reuse rights.","primary_link":"https://proceedings.mlr.press/v267/mahdavi25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DSL-Lab/aops"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jojo23333/LiveAoPSBench-2024"},{"key":"project","label":["Project","项目主页"],"url":"https://livemathbench.github.io/leaderboard.html"}],"link_count":8,"sections":9},{"id":"libra-assessing-improving-reward-model-learning-think-2025","title":"Libra: Assessing and Improving Reward Model by Learning to Think","year":2025,"venue":"arXiv","authors":["Meng Zhou","Bei Li","Jiahao Liu","Xiaowen Shi","Yang Bai","Rongxiang Weng","Jingang Wang","Xunliang Cai"],"authors_zh":"Meng Zhou、Bei Li、Jiahao Liu、Xiaowen Shi、Yang Bai、Rongxiang Weng、Jingang Wang、Xunliang Cai","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["mathematical-reasoning","reward-modeling"],"tags":["reward-model","mathematics","preference-data","reasoning","2025"],"status":"verified","priority":"可读","paper_type_zh":"数学推理奖励模型基准与训练方法","best_for_zh":"需要评测或训练能够判断复杂数学推理正确性的奖励模型的研究者","confidence":"high","one_line":["Libra Bench turns difficult mathematical reasoning outputs into a 3,740-record correctness benchmark for training and testing thinking reward models.","Libra-Bench 提供约 3,740 条数学推理偏好样本，要求奖励模型先形成评判推理再排序，直接服务于 RM 训练与推理时选择。"],"why":"It makes reward-model reasoning itself observable through labelled judgments of difficult model-generated solutions.","primary_link":"https://arxiv.org/abs/2507.21645","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/meituan/Libra-Bench"}],"link_count":3,"sections":9},{"id":"lifbench-long-context-instruction-following-2025","title":"LIFBench: Evaluating the Instruction Following Performance and Stability of Large Language Models in Long-Context Scenarios","year":2025,"venue":"ACL 2025","authors":["Xiaodong Wu","Minhao Wang","Yichen Liu","Xiaoming Shi","He Yan","Xiangju Lu","Junmin Zhu","Wei Zhang"],"authors_zh":"Xiaodong Wu、Minhao Wang、Yichen Liu、Xiaoming Shi、He Yan、Xiangju Lu、Junmin Zhu、Wei Zhang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["LIFBench uses scalable long-context instructions and LIFEval rubrics to measure instruction following and stability.","LIFBench 用可扩展长上下文指令和 LIFEval 量表，测量模型遵循指令的能力与稳定性。"],"why":"LIFBench uses scalable long-context instructions and LIFEval rubrics to measure instruction following and stability.","primary_link":"https://aclanthology.org/2025.acl-long.803/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SheldonWu0327/LIF-Bench-2024"}],"link_count":2,"sections":9},{"id":"limo-less-is-more-reasoning-2025","title":"LIMO: Less is More for Reasoning","year":2025,"venue":"COLM 2025","authors":["Yixin Ye","Zhen Huang","Yang Xiao","Ethan Chern","Shijie Xia","Pengfei Liu"],"authors_zh":"Yixin Ye、Zhen Huang、Yang Xiao、Ethan Chern、Shijie Xia、Pengfei Liu","tracks":["data_construction_open_release_recipes","instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","model_report","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematical_reasoning","competition_mathematics","multilingual_mathematics"],"tags":["limo","data-efficient-reasoning","mathematical-reasoning","long-cot","difficulty-filtering","lexical-quality-scoring","supervised-fine-tuning","teacher-distillation","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"小规模高筛选数学推理数据发布、构造 recipe 与 scaling study","best_for_zh":"研究 data-efficient reasoning SFT、难度筛选、长 CoT 质量 proxy、base-model 依赖，或审计小数据发布成本与版本一致性的读者","confidence":"high","one_line":["LIMO-v2 distills a massive mixed mathematics pool into 800 long-reasoning SFT triples using model solve-rate difficulty filters and a lexical quality score, achieving data-efficient elicitation while leaving the original pool, item lineage, rejects, and construction implementation unreleased.","LIMO-v2 用两阶段模型解题率筛选与偏好长、验证词、试探词和连接词的词法分数，从海量数学池中保留 800 条长 CoT SFT 三元组；它展示了小数据 elicitation 的可能性，但 v1/v2 漂移、构造代码与逐条 lineage 缺失使 recipe 尚不能完整重放。"],"why":"It offers a concrete, influential argument that carefully chosen reasoning demonstrations can matter more than raw SFT volume. It also shows why final row count is an incomplete measure of data efficiency: hidden cost, provenance, verifier semantics, version alignment, and rejected examples determine whether the recipe can be trusted and reproduced.","primary_link":"https://arxiv.org/abs/2502.03387","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/LIMO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GAIR/LIMO-v2"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/GAIR/LIMO-v2"}],"link_count":7,"sections":9},{"id":"limopro-reasoning-refinement-2025","title":"LIMOPro: Reasoning Refinement for Efficient and Effective Test-time Scaling","year":2025,"venue":"NeurIPS 2025","authors":["Yang Xiao","Jiashuo Wang","Ruifeng Yuan","Chunpu Xu","Kaishuai Xu","Wenjie Li","Pengfei Liu"],"authors_zh":"Yang Xiao、Jiashuo Wang、Ruifeng Yuan、Chunpu Xu、Kaishuai Xu、Wenjie Li、Pengfei Liu（机构：香港理工大学、上海交通大学、SII）","tracks":["scaling_rlvr_test_time_compute","rollout_search_test_time_trace_data"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-compute","reasoning-refinement","chain-of-thought","token-efficiency","perplexity"],"status":"verified","priority":"必读","paper_type_zh":"高效测试时推理数据与扩展研究","best_for_zh":"希望改善准确率—延迟权衡、但不愿粗暴压缩逻辑解题主线的读者。","confidence":"high","one_line":["LIMOPro prunes low-importance functional reasoning steps from training traces, producing models that retain or improve accuracy while spending fewer tokens at inference.","LIMOPro 从训练推理轨迹中剪去低重要性的功能性步骤，使模型在测试时以更少词元保持或提高准确率。"],"why":"It ties data-side step pruning to measurable inference-time scaling behavior and separates essential reasoning progress from redundant functional text.","primary_link":"https://arxiv.org/abs/2505.19187","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GAIR-NLP/LIMOPro"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/YangXiao-nlp/LIMOPro-Data-LIMO-P"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/YangXiao-nlp/LIMOPro-LIMO-P"}],"link_count":7,"sections":9},{"id":"mclm-multilingual-tts-2025","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","year":2025,"venue":"ACL 2025","authors":["Guijin Son","Jiwoo Hong","Hyunwoo Ko","James Thorne"],"authors_zh":"Guijin Son、Jiwoo Hong、Hyunwoo Ko、James Thorne（机构：Yonsei University、OneLineAI、KAIST AI、MODULABS）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","multilingual-reasoning"],"tags":["test-time-compute","multilingual","mathematical-reasoning","reward-model","budget-forcing","evaluation"],"status":"verified","priority":"必读","paper_type_zh":"多语言测试时扩展基准与比较研究（ACL 2025）","best_for_zh":"评估测试时扩展策略能否迁移到英语之外推理场景的读者。","confidence":"high","one_line":["MCLM tests outcome rewards, process rewards, and budget forcing at matched compute across 55 languages, exposing weak cross-lingual scaling gains.","MCLM 在 55 种语言中以匹配计算比较结果奖励、过程奖励和预算强制，揭示测试时扩展的跨语言收益很弱。"],"why":"It shows that a gain measured only in English can overstate the generality of an inference-time scaling method.","primary_link":"https://aclanthology.org/2025.acl-long.699/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/gauss5930/MCLM"}],"link_count":4,"sections":9},{"id":"livebench-2025","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","year":2025,"venue":"ICLR 2025 Spotlight","authors":["Colin White","Samuel Dooley","Manley Roberts","Arka Pal","Ben Feuer","Siddhartha Jain","Ravid Shwartz-Ziv","Neel Jain","Khalid Saifullah","Sreemanti Dey","Shubh-Agrawal","Sandeep Singh Sandha","Siddartha Naidu","Chinmay Hegde","Yann LeCun","Tom Goldstein","Willie Neiswanger","Micah Goldblum"],"authors_zh":"Colin White, Samuel Dooley, Manley Roberts, Arka Pal, Ben Feuer, Siddhartha Jain, Ravid Shwartz-Ziv, Neel Jain, Khalid Saifullah, Sreemanti Dey, Shubh-Agrawal, Sandeep Singh Sandha, Siddartha Naidu, Chinmay Hegde, Yann LeCun, Tom Goldstein, Willie Neiswanger, Micah Goldblum","tracks":["foundations_and_primers","benchmarks_evaluation_surfaces"],"source_role":["benchmark","survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["large-language-models","evaluation-background"],"tags":["evaluation-background","arxiv-2406.19314","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"抗污染 benchmark 与评测协议","best_for_zh":"适合设计版本化客观评测，或审计 benchmark 泄漏风险的读者。","confidence":"high","one_line":["LiveBench combines dated task releases, periodic replacement, and objective scoring so model results can be audited against a specific contamination-limited benchmark version.","LiveBench 把带日期的任务版本、定期替换和客观评分组合起来，让模型结果能够对应一个明确的抗污染 benchmark 版本接受审计。"],"why":"It makes the benchmark version and checker part of the evidence instead of treating a public test set as timeless.","primary_link":"https://arxiv.org/abs/2406.19314","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LiveBench/LiveBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/livebench"},{"key":"project","label":["Project","项目主页"],"url":"https://livebench.ai"}],"link_count":5,"sections":9},{"id":"llama-nemotron-2025","title":"Llama-Nemotron: Efficient Reasoning Models","year":2025,"venue":"arXiv preprint","authors":["NVIDIA"],"authors_zh":"NVIDIA","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","rlvr","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","code","science","instruction_following","chat","safety"],"tags":["frontier-report","nvidia","llama-nemotron","post-training-data","reasoning-toggle","synthetic-data","sft","rlvr","grpo","rloo","reward-model","disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"开放推理模型技术报告与数据披露账本","best_for_zh":"比较模型报告的发布声明与工件级可审计证据的读者","confidence":"medium","one_line":["Llama-Nemotron releases Nano, Super, and Ultra weights plus a versioned SFT/RL dataset and detailed domain pipelines, but the report's 33,011,757-example accounting is not reconciled with current released-row views or complete stage-to-record lineage.","Llama-Nemotron 报告架构搜索、蒸馏、持续预训练、推理 SFT 与大规模 RL，并声称发布模型、后训练数据和代码；具体反馈与审计边界仍需工件级核验。"],"why":"It demonstrates that frontier disclosure can include weights, data fields, domain generators, filters, and RL rewards while still leaving material gaps in release accounting, rejected candidates, verifier replay, rights lineage, and exact run composition.","primary_link":"https://arxiv.org/abs/2505.00949","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Llama-Nemotron-Post-Training-Dataset"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/nvidia/llama-nemotron"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/NVIDIA-NeMo/Nemotron"}],"link_count":5,"sections":9},{"id":"llava-cot-2025","title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","year":2025,"venue":"ICCV 2025","authors":["Guowei Xu","Peng Jin","Ziang Wu","Hao Li","Yibing Song","Lichao Sun","Li Yuan"],"authors_zh":"Guowei Xu, Peng Jin, Ziang Wu, Hao Li, Yibing Song, Lichao Sun, Li Yuan","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","visual-question-answering","mathematics","science"],"tags":["multimodal-instruction-data","structured-rationale","visual-reasoning","teacher-distillation","test-time-search"],"status":"verified","priority":"必读","paper_type_zh":"多模态推理示范数据集与结构化推理模型报告","best_for_zh":"适合构建或审计分阶段视觉推理 SFT 数据的读者。","confidence":"high","one_line":["LLaVA-CoT-100k turns 99k VQA pairs into four-stage summary-caption-reasoning-conclusion demonstrations for multimodal SFT.","LLaVA-CoT-100k 把 9.9 万条 VQA 样本改写为摘要、图像描述、推理和结论四阶段的多模态 SFT 示范。"],"why":"It exposes a compact, inspectable target schema for teaching a VLM to organize perception and reasoning before answering.","primary_link":"https://arxiv.org/abs/2411.10440","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PKU-YuanGroup/LLaVA-CoT"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Xkev/LLaVA-CoT-100k"}],"link_count":4,"sections":9},{"id":"llava-critic-learning-evaluate-multimodal-models-2025","title":"LLaVA-Critic: Learning to Evaluate Multimodal Models","year":2025,"venue":"CVPR 2025","authors":["Tianyi Xiong","Xiyao Wang","Dong Guo","Qinghao Ye","Haoqi Fan","Quanquan Gu","Heng Huang","Chunyuan Li"],"authors_zh":"Tianyi Xiong、Xiyao Wang、Dong Guo、Qinghao Ye、Haoqi Fan、Quanquan Gu、Heng Huang、Chunyuan Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","answer_level"],"training_use":["reward_modeling","preference_learning","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["multimodal-evaluation","vision-language"],"tags":["vision-language","critic","preference-data","cvpr-2025"],"status":"verified","priority":"可读","paper_type_zh":"多模态通用评审模型与 critic 指令数据集","best_for_zh":"需要视觉语言 judge、带理由的评估监督或多模态偏好训练数据的研究者","confidence":"high","one_line":["LLaVA-Critic releases 113,000 multimodal evaluation records—scores, preference pairs, and rationales—to train an open generalist vision-language evaluator.","LLaVA-Critic-113k 含 46K 图像、113K 评分或成对偏好及理由，覆盖多种视觉任务，可用于多模态 judge 与偏好训练。"],"why":"It combines criterion-following scoring, pairwise preference, and rationale generation in one open multimodal critic dataset.","primary_link":"https://openaccess.thecvf.com/content/CVPR2025/html/Xiong_LLaVA-Critic_Learning_to_Evaluate_Multimodal_Models_CVPR_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/llava-vl/LLaVA-Critic"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lmms-lab/llava-critic-113k"},{"key":"project","label":["Project","项目主页"],"url":"https://llava-vl.github.io/blog/2024-10-03-llava-critic"}],"link_count":5,"sections":9},{"id":"llava-onevision-15-2025","title":"LLaVA-OneVision-1.5: Fully Open Framework for Democratized Multimodal Training","year":2025,"venue":"arXiv preprint","authors":["An, Xiang","Xie, Yin","Yang, Kaicheng","Zhang, Wenkang","Zhao, Xiuwei","Cheng, Zheng","Wang, Yirui","Xu, Songcen","Chen, Changrui","Wu, Chunsheng","Huajie Tan","Li, Chunyuan","Jing Yang","Jie Yu","Xiyao Wang","Bin Qin","Yumeng Wang","Zizhen Yan","Ziyong Feng","Ziwei Liu","Bo Li","Jiankang Deng"],"authors_zh":"An, Xiang、Xie, Yin、Yang, Kaicheng、Zhang, Wenkang、Zhao, Xiuwei、Cheng, Zheng、Wang, Yirui、Xu, Songcen、Chen, Changrui、Wu, Chunsheng、Huajie Tan、Li, Chunyuan、Jing Yang、Jie Yu、Xiyao Wang、Bin Qin、Yumeng Wang、Zizhen Yan、Ziyong Feng、Ziwei Liu、Bo Li、Jiankang Deng","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["fully-disclosed-large-scale-multimodal-instruction-mixture"],"tags":["instruction-demonstration-rationale","arxiv-2509.23661","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"可选强化学习之前的全模型多模态监督微调","confidence":"high","one_line":["LLaVA-OneVision-1.5 releases a 22M instruction mixture and an efficient full training stack, separating the instruction stage from its 85M mid-training corpus and 67K RL set.","LLaVA-OneVision-1.5 公开 2200 万条多模态指令混合及完整训练框架，并把它与中期训练和强化学习数据分开。"],"why":"Open multimodal recipes often publish weights without the exact large instruction mixture, preventing controlled study of data balance and economical training.","primary_link":"https://arxiv.org/abs/2509.23661","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mvp-lab/LLaVA-OneVision-1.5-Instruct-Data"}],"link_count":2,"sections":9},{"id":"llm-judge-extractive-qa-2025","title":"LLM-as-a-Judge: Reassessing the Performance of LLMs in Extractive QA","year":2025,"venue":"arXiv preprint","authors":["Xanh Ho","Jiahao Huang","Florian Boudin","Akiko Aizawa"],"authors_zh":"Xanh Ho, Jiahao Huang, Florian Boudin, Akiko Aizawa","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["llm-as-a-judge","extractive-qa","audit"],"status":"verified","priority":"可读","paper_type_zh":"大语言模型评审可靠性实证研究","best_for_zh":"需要核查问答自动评测可靠性的研究者。","confidence":"medium","one_line":["Audits LLM judging against human labels for extractive QA where EM and F1 can miss valid paraphrases.","审计大语言模型评审能否修正 EM 与 F1 对抽取式问答有效改述的误判。"],"why":"It provides a reproducible reliability test for replacing lexical QA metrics with contextual judgments.","primary_link":"https://arxiv.org/abs/2504.11972","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Alab-NII/llm-judge-extract-qa"}],"link_count":2,"sections":9},{"id":"planning-formalizers-survey-2025","title":"LLMs as Planning Formalizers: A Survey for Leveraging Large Language Models to Construct Automated Planning Models","year":2025,"venue":"Findings of ACL 2025","authors":["Marcus Tantakoun","Christian Muise","Xiaodan Zhu"],"authors_zh":"Marcus Tantakoun 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["planning","structured-reasoning","neuro-symbolic"],"tags":["foundations-and-primers","planning","structured-reasoning","neuro-symbolic","acl-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"规划形式化综述","best_for_zh":"探索大语言模型如何辅助长程规划与形式化规划模型构建的读者。","confidence":"high","one_line":["A Findings ACL 2025 survey of using language models to formalize planning specifications for automated planners.","综述如何借助大语言模型构造和改进自动规划模型。"],"why":"It clarifies the boundary between generating a plan in prose and constructing a formal model that a planner can reliably use.","primary_link":"https://aclanthology.org/2025.findings-acl.1291/","links":[],"link_count":2,"sections":9},{"id":"robustjudge-2025","title":"LLMs Cannot Reliably Judge (Yet?): A Comprehensive Assessment on the Robustness of LLM-as-a-Judge","year":2025,"venue":"arXiv preprint","authors":["Songze Li","Chuokun Xu","Jiaying Wang","Xueluan Gong","Chen Chen","Jirui Zhang","Jun Wang","Kwok-Yan Lam","Shouling Ji"],"authors_zh":"Songze Li, Chuokun Xu, Jiaying Wang, Xueluan Gong, Chen Chen, Jirui Zhang, Jun Wang, Kwok-Yan Lam, Shouling Ji","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Public RobustJudge release evaluates reliability under multiple automated judge-attack scenarios.","以 15 类攻击和 7 类防御审计 LLM-as-a-Judge 的稳健性。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://arxiv.org/abs/2506.09443","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/S3IC-Lab/RobustJudge"}],"link_count":2,"sections":9},{"id":"judge-bench-llms-instead-human-judges-2025","title":"LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks","year":2025,"venue":"ACL 2025","authors":["Anna Bavaresco","Raffaella Bernardi","Leonardo Bertolazzi","Desmond Elliott","Raquel Fern?ndez","Albert Gatt","Esam Ghaleb","Mario Giulianelli","Michael Hanna","Alexander Koller","Andre Martins","Philipp Mondorf","Vera Neplenbroek","Sandro Pezzelle","Barbara Plank","David Schlangen","Alessandro Suglia","Aditya K Surikuchi","Ece Takmaz","Alberto Testoni"],"authors_zh":"Anna Bavaresco、Raffaella Bernardi、Leonardo Bertolazzi、Desmond Elliott、Raquel Fern?ndez、Albert Gatt、Esam Ghaleb、Mario Giulianelli、Michael Hanna、Alexander Koller、Andre Martins、Philipp Mondorf、Vera Neplenbroek、Sandro Pezzelle、Barbara Plank、David Schlangen、Alessandro Suglia、Aditya K Surikuchi、Ece Takmaz、Alberto Testoni","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward","pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","evaluation"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"必读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["JUDGE-BENCH is the paper's released feedback or evaluation resource.","JUDGE-BENCH 汇聚 20 个带人类评分的数据集，覆盖事实性、推理、对话和生成质量，用于定位 LLM judge 与专家判断偏离的条件。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2025.acl-short.20/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/dmg-illc/JUDGE-BENCH"}],"link_count":3,"sections":9},{"id":"llmsql-wikisql-modern-text-to-sql-2025","title":"LLMSQL: Upgrading WikiSQL for the LLM Era of Text-to-SQL","year":2025,"venue":"2025 IEEE International Conference on Data Mining Workshops (ICDMW)","authors":["Dzmitry Pihulski","Karol Charchut","Viktoria Novogrodskaia","Jan Kocoń"],"authors_zh":"Dzmitry Pihulski, Karol Charchut, Viktoria Novogrodskaia, Jan Kocoń","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code","database"],"tags":["text-to-sql","wikisql","sqlite","execution","icdmw-2025"],"status":"verified","priority":"可读","paper_type_zh":"清洗并可执行验证的单表 Text-to-SQL 基准","best_for_zh":"需要现代 Text-to-SQL 单表基线或数据清洗范式的研究者。","confidence":"high","one_line":["LLMSQL repairs and re-releases WikiSQL as complete SQL records scored by executable SQLite result sets.","LLMSQL 修复并重发布 WikiSQL 为完整 SQL 记录，再以可执行的 SQLite 结果集合评分。"],"why":"It makes a legacy Text-to-SQL dataset usable for modern full-SQL generation while preserving programmatic checking.","primary_link":"https://arxiv.org/abs/2510.02350","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/llmsql-bench/llmsql"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/llmsql-bench/llmsql-benchmark"},{"key":"project","label":["Project","项目主页"],"url":"https://llmsql.github.io/llmsql-benchmark/"}],"link_count":5,"sections":9},{"id":"lmarena-leaderboard-2025","title":"LMArena / Arena AI Leaderboard","year":2025,"venue":"Arena AI / LMArena leaderboard infrastructure","authors":["Arena AI","LMArena"],"authors_zh":"Arena AI、LMArena","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["model-evaluation","human-preference","leaderboard"],"tags":["benchmark","evaluation-surface","leaderboard","pairwise_preference"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"关注开放式对话偏好评测、排行榜基础设施、模型排名方法和人类投票偏差的读者。","confidence":"high","one_line":["LMArena is a continuing human-preference leaderboard infrastructure that ranks models from anonymized pairwise battles rather than fixed-answer correctness.","LMArena 是持续运行的人类偏好排行榜基础设施，通过匿名双模型对战和用户投票聚合模型排名，而不是固定答案正确率。"],"why":"LMArena is a continuing human-preference leaderboard infrastructure that ranks models from anonymized pairwise battles rather than fixed-answer correctness.","primary_link":"https://arena.ai/leaderboard","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lmarena/arena-rank"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/lmarena-ai"}],"link_count":3,"sections":9},{"id":"lne-blocking-2025","title":"LNE-Blocking: An Efficient Framework for Contamination Mitigation Evaluation on Large Language Models","year":2025,"venue":"Findings of EMNLP 2025","authors":[],"authors_zh":"Ruijie Hou, Yueyang Jiao, Hanxu Hu, Yingming Li, Wai Lam, Huajian Zhang, Hongyuan Lu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["contamination-audit","evaluation-reliability","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"Findings of EMNLP 2025 的基准污染缓解与评测可靠性研究","best_for_zh":"需要审计污染基准、比较记忆与泛化行为或设计污染缓解评测的研究者。","confidence":"high","one_line":["contamination detection and response-blocking evaluation code","LNE-Blocking 以长度归一化熵决定解码扰动强度，估计潜在泄漏基准上的非记忆性能。"],"why":"It offers an inference-time audit for benchmark contamination and evaluation reliability.","primary_link":"https://aclanthology.org/2025.findings-emnlp.188/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RuijieH/LNE-Blocking"}],"link_count":2,"sections":9},{"id":"logic-orm-test-time-scaling-2025","title":"Logical Reasoning with Outcome Reward Models for Test-Time Scaling","year":2025,"venue":"EMNLP 2025","authors":["Ramya Keerthy Thatikonda","Wray Buntine","Ehsan Shareghi"],"authors_zh":"Ramya Keerthy Thatikonda、Wray Buntine、Ehsan Shareghi","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer"],"domains":["deductive-logic","natural-language-reasoning","logical-reasoning"],"tags":["outcome-reward-model","best-of-n","test-time-scaling","echo-cot","hard-negative-mining","deductive-logic","folio","proverqa","justlogic","qwen2.5","gpt-4o","rejected-candidates"],"status":"partial","priority":"必读","paper_type_zh":"Outcome Reward Model 数据发布与 Best-of-N 测试时扩展研究","best_for_zh":"研究多候选推理、结果奖励建模、hard-negative 构造、verifier 审计与测试时算力分配的读者","confidence":"high","one_line":["LogicORM releases 10,009 FOLIO CoT traces and a 19,105-row Echo-augmented set with binary outcome labels, then trains Qwen2.5 ORMs to rerank sampled deductive-logic solutions at test time.","LogicORM 公开 10,009 条 FOLIO CoT 轨迹及含 19,105 条记录的 Echo 增强集，以二元结果标签训练 Qwen2.5 ORM，并在测试时对多次采样的演绎逻辑解答执行 Best-of-N 重排。"],"why":"It ties candidate-generation budgets, positive and negative trace distributions, echo-based hard-negative mining, scalar ORM supervision, and best-of-N selection into one pipeline while exposing consequential gaps in provenance, rejected traces, score logs, splits, and verifier validity.","primary_link":"https://aclanthology.org/2025.emnlp-main.1326/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RamyaKeerthy/LogicORM"},{"key":"data","label":["Data","数据"],"url":"https://github.com/RamyaKeerthy/LogicORM/tree/main/data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/ramyakeerthyt/qwen25-logic-orm"}],"link_count":6,"sections":9},{"id":"ls-mixture-efficient-reasoning-2025","title":"Long-Short Chain-of-Thought Mixture Supervised Fine-Tuning Eliciting Efficient Reasoning in Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Bin Yu","Hang Yuan","Haotian Li","Xueyin Xu","Yuliang Wei","Bailing Wang","Weizhen Qi","Kai Chen"],"authors_zh":"Bin Yu, Hang Yuan, Haotian Li, Xueyin Xu, Yuliang Wei, Bailing Wang, Weizhen Qi, Kai Chen","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["reasoning","instruction-tuning"],"tags":["post-training","training-usage","reasoning-data"],"status":"verified","priority":"必读","paper_type_zh":"推理数据构造与监督微调论文","best_for_zh":"研究推理记录如何进入后训练的读者。","confidence":"high","one_line":["LS-Mixture SFT pairs long reasoning traces with structure-preserving short rewrites to train efficient reasoning.","LS-Mixture SFT 将长推理轨迹与保留结构的短改写配对，以训练更高效的推理。"],"why":"It makes a reasoning-data consumption decision explicit.","primary_link":"https://arxiv.org/abs/2505.03469","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ZGCA-AI4Edu/LS-Mixture"}],"link_count":2,"sections":9},{"id":"longdpo-stepwise-longform-preference-2025","title":"LongDPO: Unlock Better Long-form Generation Abilities for LLMs via Critique-augmented Stepwise Information","year":2025,"venue":"Findings of ACL 2025","authors":["Bowen Ping","Jiali Zeng","Fandong Meng","Shuo Wang","Jie Zhou","Shanghang Zhang"],"authors_zh":"Bowen Ping, Jiali Zeng, Fandong Meng, Shuo Wang, Jie Zhou, Shanghang Zhang（北京大学、腾讯微信 AI、清华大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["long_form_generation","alignment","preference_learning"],"tags":["long-form-generation","preference-data","mcts","critique","stepwise-dpo"],"status":"verified","priority":"必读","paper_type_zh":"逐步偏好数据构造与长文本对齐研究","best_for_zh":"适合构建长文本生成对齐流程、而完整答案偏好无法定位事实或结构失败位置的读者。","confidence":"high","one_line":["LongDPO uses MCTS, factual memory, and external critiques to build stepwise preference pairs for fine-grained long-form DPO.","LongDPO 用蒙特卡洛树搜索、事实记忆和外部批评构造逐步偏好回答对，供细粒度长文本 DPO 使用。"],"why":"It turns long-form training from an outcome-only preference problem into a sequence of selected and refined local decisions.","primary_link":"https://aclanthology.org/2025.findings-acl.395/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pingbowen23/LongDPO"}],"link_count":3,"sections":9},{"id":"m-mad-llm-judge-2025","title":"M-MAD: Multidimensional Multi-Agent Debate for Advanced Machine Translation Evaluation","year":2025,"venue":"ACL 2025","authors":["Zhaopeng Feng","Jiayuan Su","Jiamei Zheng","Jiahan Ren","Yan Zhang","Jian Wu","Hongwei Wang","Zuozhu Liu"],"authors_zh":"Zhaopeng Feng, Jiayuan Su, Jiamei Zheng, Jiahan Ren, Yan Zhang, Jian Wu, Hongwei Wang, Zuozhu Liu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"自动评审可靠性与 LLM-as-a-judge 评测论文","best_for_zh":"需要构建或审计机器翻译 LLM 评审器的研究者。","confidence":"high","one_line":["multi-judge debate data and evaluator reproducibility assets","将 MQM 维度拆分、维度内辩论与最终综合结合，用于细粒度机器翻译评测。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://aclanthology.org/2025.acl-long.351/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SU-JIAYUAN/M-MAD"}],"link_count":2,"sections":9},{"id":"m-rewardbench-multilingual-settings-2025","title":"M-RewardBench: Evaluating Reward Models in Multilingual Settings","year":2025,"venue":"ACL 2025","authors":["Srishti Gureja","Lester James V. Miranda","Shayekh Bin Islam","Rishabh Maheshwary","Drishti Sharma","Gusti Winata","Nathan Lambert","Sebastian Ruder","Sara Hooker","Marzieh Fadaee"],"authors_zh":"Srishti Gureja, Lester James V. Miranda, Shayekh Bin Islam, Rishabh Maheshwary, Drishti Sharma, Gusti Winata, Nathan Lambert, Sebastian Ruder, Sara Hooker, Marzieh Fadaee","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multilingual","reward-modeling","reasoning-evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多语言奖励模型评测与偏好数据基准论文","best_for_zh":"需要检验奖励模型在多语言对话、安全、推理和翻译任务中是否保持稳定偏好的研究者。","confidence":"high","one_line":["M-RewardBench translates and validates preference pairs across 23 languages to test whether reward models preserve chat, safety, reasoning, and translation preferences.","2.87k偏好实例覆盖23种语言及推理子集，用成对优劣标签检验跨语言通用推理奖励。"],"why":"It exposes language-specific feedback failures that an English-only reward benchmark cannot reveal.","primary_link":"https://aclanthology.org/2025.acl-long.3/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/for-ai/m-rewardbench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/C4AI-Community/multilingual-reward-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://m-rewardbench.github.io/"}],"link_count":6,"sections":9},{"id":"macosworld-2025","title":"macOSWorld: A Multilingual Interactive Benchmark for GUI Agents","year":2025,"venue":"NeurIPS 2025 (Main Conference Track)","authors":["Pei Yang","Hai Ci","Mike Zheng Shou"],"authors_zh":"Pei Yang, Hai Ci, Mike Zheng Shou","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","gui_agents","multilingual","safety"],"tags":["environment-agent-trajectory-data","gui-agent","interactive-benchmark","multilingual","programmatic-evaluation","context-deception"],"status":"verified","priority":"可读","paper_type_zh":"多语言GUI智能体交互基准与可执行环境","best_for_zh":"研究环境交互轨迹、GUI智能体评测、程序化反馈与上下文欺骗安全审计的读者","confidence":"high","one_line":["macOSWorld releases 202 resettable macOS GUI tasks with multilingual instructions, VNC screenshot/action episodes, task-specific executable graders, and an overlapping 29-task context-deception safety subset.","macOSWorld发布202项可重置的多语言macOS GUI任务、可执行终态grader及重叠的29项上下文欺骗安全子集，但公开grader、二值反馈和缺失的固定轨迹清单限制训练复用。"],"why":"It makes the environment and feedback contract auditable at task level—prompt, snapshot, preparation, action surface, terminal predicate, and safety event—while revealing that binary evaluators, public graders, cloud/version drift, and missing rollout/split provenance constrain training reuse.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/c276c3303c0723c83a43b95a44a1fcbf-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/showlab/macosworld"},{"key":"data","label":["Data","数据"],"url":"https://github.com/showlab/macosworld/tree/master/tasks"},{"key":"project","label":["Project","项目主页"],"url":"https://macos-world.github.io/"}],"link_count":8,"sections":9},{"id":"magistral-2025","title":"Magistral","year":2025,"venue":"arXiv preprint","authors":["Mistral AI"],"authors_zh":"Mistral AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","distillation","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","coding","multilingual_reasoning"],"tags":["magistral","frontier-report","data-disclosure-ledger","grpo","rlvr"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型技术报告与数据披露账本","best_for_zh":"审计 RLVR、程序化验证和推理 trace 冷启动边界的读者","confidence":"high","one_line":["Magistral reports filtered math/code RL and Medium-to-Small trace transfer, but releases only Small model weights rather than the training corpus, traces, tests or reward logs.","Magistral 具体披露数学/代码筛选、结果奖励、异步 GRPO 和 Medium-to-Small 冷启动，但未发布来源清单、trace、完整测试和审计证据。"],"why":"It shows how to audit a technically detailed frontier RL disclosure without mistaking a model-weight release or benchmark gain for a reusable data release.","primary_link":"https://arxiv.org/abs/2506.10910","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/mistralai/Magistral-Small-2506"},{"key":"project","label":["Project","项目主页"],"url":"https://mistral.ai/news/magistral/"}],"link_count":5,"sections":9},{"id":"magnet-tool-use-2025","title":"Magnet: Multi-turn Tool-use Data Synthesis and Distillation via Graph Translation","year":2025,"venue":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","authors":["Fan Yin","Zifeng Wang","I-Hung Hsu","Jun Yan","Ke Jiang","Yanfei Chen","Jindong Gu","Long T. Le","Kai-Wei Chang","Chen-Yu Lee","Hamid Palangi","Tomas Pfister"],"authors_zh":"Fan Yin、Zifeng Wang、I-Hung Hsu、Jun Yan、Ke Jiang、Yanfei Chen、Jindong Gu、Long T. Le、Kai-Wei Chang、Chen-Yu Lee、Hamid Palangi、Tomas Pfister","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","pairwise_preference"],"training_use":["sft","preference_learning","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","optimizer_scaffold"],"domains":["tool_use","function_calling","multi_turn_agents"],"tags":["magnet","tool-use","function-calling","multi-turn-agent","synthetic-trajectories","graph-translation","context-distillation","mdpo"],"status":"partial","priority":"可读","paper_type_zh":"多轮函数调用轨迹合成、上下文蒸馏与偏好优化研究","best_for_zh":"研究工具调用 agent 数据构造、函数调用环境、轨迹偏好对与可复现性审计的读者","confidence":"medium","one_line":["MAGNET synthesizes graph-derived multi-turn function-call episodes and positive/negative trajectory pairs for SFT and mDPO, but the paper does not release the traces, executable environment, or replay metadata needed to audit the recipe end to end.","MAGNET 从 API 依赖图合成多轮函数调用轨迹，并构造正负轨迹对用于 SFT 与 mDPO；但未发布轨迹、可执行环境或回放元数据，无法端到端审计该配方。"],"why":"It makes the data-construction interface unusually explicit: function dependencies, calls, outputs, teacher hints, SFT errors, and preference pairs are separate objects. The absence of released episodes, API versions, prompts-as-executed, licenses, and full verifier audits prevents treating the published scores as evidence of a reusable dataset.","primary_link":"https://aclanthology.org/2025.acl-long.1566/","links":[],"link_count":3,"sections":9},{"id":"magpie-2025","title":"Magpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing","year":2025,"venue":"ICLR 2025","authors":["Zhangchen Xu","Fengqing Jiang","Luyao Niu","Yuntian Deng","Radha Poovendran","Yejin Choi","Bill Yuchen Lin"],"authors_zh":"Zhangchen Xu, Fengqing Jiang, Luyao Niu, Yuntian Deng, Radha Poovendran, Yejin Choi, Bill Yuchen Lin","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["general-instruction-following","synthetic-data","multilingual","mathematics","code"],"tags":["synthetic-instruction-data","self-synthesis","alignment-data","llama-3","arxiv-2406.08464"],"status":"verified","priority":"必读","paper_type_zh":"合成指令数据集与自动构造流程","best_for_zh":"适合研究大规模合成指令、教师分布继承与 SFT 数据筛选的构建者和审计者。","confidence":"high","one_line":["Magpie samples seedless instructions from an aligned model's chat-template prefix and releases millions of instruction-response demonstrations plus filtered SFT subsets.","Magpie 只向已对齐模型输入聊天模板中的用户角色前缀，就能无种子地采样指令与回答，并开放数百万条示范及其筛选版 SFT 数据。"],"why":"It makes large-scale instruction generation possible without seed questions or commercial teacher APIs while exposing how teacher choice and filtering shape the resulting training distribution.","primary_link":"https://arxiv.org/abs/2406.08464","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/magpie-align/magpie"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Magpie-Align/Magpie-Pro-300K-Filtered"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Magpie-Align"},{"key":"project","label":["Project","项目主页"],"url":"https://magpie-align.github.io/"}],"link_count":7,"sections":9},{"id":"majority-bests-2025","title":"Majority of the Bests: Improving Best-of-N via Bootstrapping","year":2025,"venue":"NeurIPS 2025","authors":["Amin Rakhsha","Kanika Madan","Tianyu Zhang","Amir-massoud Farahmand","Amir Khasahmadi"],"authors_zh":"Amin Rakhsha、Kanika Madan、Tianyu Zhang、Amir-massoud Farahmand、Amir Khasahmadi","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["test_time_compute"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["mathematics","science","commonsense_reasoning"],"tags":["majority-of-the-bests","mob","best-of-n","bootstrapping","reward-model-selection","test-time-compute"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["MoB estimates the modal answer of a reward-ranked Best-of-m distribution from a fixed completion pool and releases prompts, generations, extracted answers, correctness scores, and reward scores for offline selection studies.","MoB 从固定的候选池中估计奖励排序后 Best-of-m 分布的众数答案，并发布提示、生成结果、抽取答案、正确性与奖励分数以支持离线选择研究。"],"why":"It exposes a reusable test-time selection object for Track 5 and separates generation budget, reward ranking, answer aggregation, and final benchmark correctness, while provenance and data-rights questions remain.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/36556567e8437f137da23047309155dd-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/arakhsha/mob"},{"key":"data","label":["Data","数据"],"url":"https://github.com/arakhsha/mob/tree/main/data"}],"link_count":7,"sections":9},{"id":"mammoth-vl-2025","title":"MAmmoTH-VL: Eliciting Multimodal Reasoning with Instruction Tuning at Scale","year":2025,"venue":"ACL 2025 Main","authors":["Jarvis Guo","Tuney Zheng","Yuelin Bai","Bo Li","Yubo Wang","King Zhu","Yizhi Li","Graham Neubig","Wenhu Chen","Xiang Yue"],"authors_zh":"Jarvis Guo, Tuney Zheng, Yuelin Bai, Bo Li, Yubo Wang, King Zhu, Yizhi Li, Graham Neubig, Wenhu Chen, Xiang Yue","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","visual-question-answering","mathematics","charts","ocr"],"tags":["multimodal-instruction-data","rationale-rewriting","open-model-distillation","visual-data-filtering","large-scale-sft"],"status":"verified","priority":"必读","paper_type_zh":"大规模多模态 rationale 数据集与构造配方","best_for_zh":"适合用开放教师构建或审计大规模多模态推理 SFT 混合数据的读者。","confidence":"high","one_line":["MAmmoTH-VL rewrites and filters 153 public multimodal sources into 12M rationale-rich instruction-response records using open models.","MAmmoTH-VL 用开源模型把 153 个公开多模态来源改写并过滤为 1200 万条带 rationale 的 instruction-response 记录。"],"why":"It demonstrates a fully open-model path from heterogeneous visual corpora to a training-scale reasoning mixture with explicit filtering ablations.","primary_link":"https://arxiv.org/abs/2412.05237","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MAmmoTH-VL/MAmmoTH-VL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MAmmoTH-VL/MAmmoTH-VL-Instruct-12M"},{"key":"project","label":["Project","项目主页"],"url":"https://mammoth-vl.github.io/"}],"link_count":7,"sections":9},{"id":"marco-o1-v2-2025","title":"Marco-o1 v2: Towards Widening The Distillation Bottleneck for Reasoning Models","year":2025,"venue":"ACL 2025","authors":["Huifeng Yin","Yu Zhao","Minghao Wu","Xuanfan Ni","Bo Zeng","Hao Wang","Tianqi Shi","Liangying Shao","Chenyang Lyu","Longyue Wang","Weihua Luo","Kaifu Zhang"],"authors_zh":"Huifeng Yin；Yu Zhao；Minghao Wu；Xuanfan Ni；Bo Zeng；Hao Wang；Tianqi Shi；Liangying Shao；Chenyang Lyu；Longyue Wang；Weihua Luo；Kaifu Zhang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["step_level","pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["search_substrate","trace_writing","optimizer_scaffold"],"domains":["reasoning"],"tags":["mcts","chain-of-thought","sft","dpo","open-ended-reasoning"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["An open reasoning pipeline connecting MCTS chain-of-thought search to SFT and DPO data construction.","Marco-o1 v2 从头用 MCTS 构造树状 CoT 数据，再结合思考长度平衡、细粒度 DPO 和联合后训练目标缓解长推理蒸馏瓶颈。"],"why":"It is a direct example of multi-stage search trace reuse; audit depends on retaining the search tree and preference-pair derivation, not only final demonstrations.","primary_link":"https://aclanthology.org/2025.acl-long.1145/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AIDC-AI/Marco-o1"}],"link_count":4,"sections":9},{"id":"massive-sft-factors-2025","title":"Massive Supervised Fine-tuning Experiments Reveal How Data, Layer, and Training Factors Shape LLM Alignment Quality","year":2025,"venue":"EMNLP 2025","authors":["Yuto Harada","Yusuke Yamauchi","Yusuke Oda","Yohei Oseki","Yusuke Miyao","Yu Takagi"],"authors_zh":"Yuto Harada, Yusuke Yamauchi, Yusuke Oda, Yohei Oseki, Yusuke Miyao, Yu Takagi","tracks":["training_usage_optimization_objectives"],"source_role":["scaling_study","audit_failure"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation","audit"],"construction_layer":["scaling_report","release_audit"],"domains":["instruction-tuning","mathematics","code","multilingual"],"tags":["sft","data-selection","perplexity","training-audit"],"status":"verified","priority":"必读","paper_type_zh":"大规模监督微调数据与训练因素审计","best_for_zh":"跨基础模型和任务验证监督微调数据选择规则的读者。","confidence":"high","one_line":["A controlled release of 1,000+ SFT runs finds base-model perplexity is a strong compatibility signal for choosing training data.","超过 1,000 次受控监督微调表明，基础模型困惑度是选择训练数据的强兼容性信号。"],"why":"It shows that the data selection signal must be tied to the base model and downstream contract.","primary_link":"https://aclanthology.org/2025.emnlp-main.1138/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/llm-jp/massive-sft"}],"link_count":4,"sections":9},{"id":"matharena-2025","title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks platform / arXiv","authors":["Mislav Balunović","Jasper Dekoninck","Ivo Petrov","Nikola Jovanović","Martin Vechev"],"authors_zh":"Mislav Balunović 等（ETH Zurich）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit","reward_verifier_layer"],"domains":["live-math","live-hidden-contamination-audit"],"tags":["benchmark","live_hidden_contamination_audit","live-math"],"status":"verified","priority":"必读","paper_type_zh":"NeurIPS 2025 Datasets and Benchmarks platform / arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MathArena exposes newly released math competitions and current contest math as an auditable evaluation surface.","MathArena 把新发布的数学竞赛与当期赛题做成可审计的评测面。"],"why":"Good live-math lead, but exact paper/source and scoring policy need confirmation.","primary_link":"https://arxiv.org/abs/2505.23281","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/eth-sri/matharena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/MathArena"}],"link_count":5,"sections":9},{"id":"mathcoder2-2025","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","year":2025,"venue":"ICLR 2025 Spotlight","authors":["Zimu Lu","Aojun Zhou","Ke Wang","Houxing Ren","Weikang Shi","Junting Pan","Mingjie Zhan","Hongsheng Li"],"authors_zh":"Zimu Lu、Aojun Zhou、Ke Wang、Houxing Ren、Weikang Shi、Junting Pan、Mingjie Zhan、Hongsheng Li","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning","code_reasoning","continued_pretraining","tool_integrated_reasoning"],"tags":["mathcoder2","mathcode-pile","mathematical-reasoning","code-reasoning","continued-pretraining","model-translated-code","execution-filtering","partial-data-release"],"status":"partial","priority":"必读","paper_type_zh":"数学 continued-pretraining 数据构建、执行过滤与开放发布审计","best_for_zh":"研究模型翻译的数学代码、continued pretraining、程序执行筛选、混合数据配方，以及 partial release 的 sandbox、谱系、许可、去污染和版本绑定风险","confidence":"medium","one_line":["MathCoder2 translates selected mathematical documents into conditions-expression-result-Python records, execution-filters and flattens them into MathCode-Pile, then uses the mixture for continued pretraining before separate supervised math post-training.","MathCoder2 将筛选后的数学网页翻译为“条件—表达式—结果—Python”记录，与网页、合成、代码和教材数据组成论文报告的 19,487,652 文档、19,184,073,343 token 的 MathCode-Pile；当前公开数据仍是只有 train split 的 partial text-only 版本，且执行过滤、谱系、许可、去污染和 checkpoint 绑定均不完整。"],"why":"It shows executable mathematical-code translation as a continued-pretraining recipe while exposing why partial data, self-consistency verification, weak execution sandboxing, broken code paths, and missing provenance must be audited separately from paper-scale results.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/file/bea94fe9c5573e74294657f692069d89-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mathllm/MathCoder2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MathGenie/MathCode-Pile"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/MathGenie/mathcoder2"},{"key":"project","label":["Project","项目主页"],"url":"https://mathllm.github.io/mathcoder2/"}],"link_count":13,"sections":9},{"id":"mathconstruct-constructive-proofs-2025","title":"MathConstruct: Challenging LLM Reasoning with Constructive Proofs","year":2025,"venue":"ICML 2025","authors":["Mislav Balunović","Jasper Dekoninck","Nikola Jovanović","Ivo Petrov","Martin Vechev"],"authors_zh":"Mislav Balunović, Jasper Dekoninck, Nikola Jovanović, Ivo Petrov, Martin Vechev","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["mathematical-reasoning","constructive-proofs","programmatic-evaluation"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"构造性数学证明与程序化检查基准","best_for_zh":"需要测试构造型数学推理、答案检查器或鲁棒性变体评测的研究者。","confidence":"high","one_line":["MathConstruct releases 126 constructive mathematics problems with programmatic checkers and generated variations for robustness evaluation.","MathConstruct 提供 126 道构造型竞赛题与程序化检查器，并通过题目变体检验模型是否真正满足构造约束。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://arxiv.org/abs/2502.10197","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/eth-sri/mathconstruct"},{"key":"data","label":["Data","数据"],"url":"https://github.com/eth-sri/mathconstruct/tree/main/data"}],"link_count":3,"sections":9},{"id":"mcts-judge-test-time-scaling-2025","title":"MCTS-Judge: Test-Time Scaling in LLM-as-a-Judge for Code Correctness Evaluation","year":2025,"venue":"arXiv preprint","authors":["Yutong Wang","Pengliang Ji","Chaoqun Yang","Kaixin Li","Ming Hu","Jiaoyang Li","Guillaume Sartoretti"],"authors_zh":"Yutong Wang；Pengliang Ji；Chaoqun Yang；Kaixin Li；Ming Hu；Jiaoyang Li；Guillaume Sartoretti","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["state_action_level","trajectory_value"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["code","reasoning"],"tags":["mcts","llm-judge","test-generation","simulated-execution","test-time-scaling"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A test-time MCTS framework for code-correctness judging that exposes trajectory-aware self-assessment and unit-test-level reward as its feedback objects.","MCTS-Judge 将代码正确性判断分解为树搜索，结合历史动作自评、UCT 和单元测试级奖励机制形成测试时 LLM-as-a-Judge 轨迹。"],"why":"It is an instructive high-risk trace source: MCTS trajectories, self-assessment, UCT values, and executable reward evidence must be preserved before LLM judging can be treated as reusable feedback.","primary_link":"https://arxiv.org/abs/2502.12468","links":[],"link_count":1,"sections":9},{"id":"mcts-rag-2025","title":"MCTS-RAG: Enhancing Retrieval-Augmented Generation with Monte Carlo Tree Search","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Yunhai Hu","Yilun Zhao","Chen Zhao","Arman Cohan"],"authors_zh":"Yunhai Hu；Yilun Zhao；Chen Zhao；Arman Cohan","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","trajectory_value","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","scaling_report"],"domains":["multi_hop_question_answering","science_question_answering","fact_checking","knowledge_intensive_reasoning"],"tags":["mcts","retrieval_augmented_generation","adaptive_retrieval","inference_time_search","rollout_scaling","trajectory_selection","self_consistency_reward","search_trace_release_gap","retrieval_lineage","test_time_compute"],"status":"partial","priority":"必读","paper_type_zh":"检索增强的测试时 MCTS 构造配方与扩展研究","best_for_zh":"研究搜索轨迹、adaptive retrieval、测试时计算与发布完整性审计的读者","confidence":"medium","one_line":["MCTS-RAG interleaves six reasoning and retrieval actions in UCT-guided search, but releases code and benchmark inputs rather than the paper-run trees, score traces, retrieval states, or budget logs.","MCTS-RAG 用 UCT 引导的搜索交替执行六类推理与检索动作，但官方发布的是生成器代码和少量 benchmark 输入，而不是论文运行产生的树、检索状态、选择分数或逐样本预算日志。"],"why":"It makes adaptive retrieval a branch-level test-time decision and quantifies rollout-cost scaling, while showing why a generator repository is not equivalent to an auditable release of generated search trajectories.","primary_link":"https://aclanthology.org/2025.findings-emnlp.672/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yale-nlp/MCTS-RAG"},{"key":"data","label":["Data","数据"],"url":"https://github.com/yale-nlp/MCTS-RAG/tree/main/data"}],"link_count":5,"sections":9},{"id":"faithful-thinking-drafts-lrm-2025","title":"Measuring the Faithfulness of Thinking Drafts in Large Reasoning Models","year":2025,"venue":"arXiv preprint","authors":["Zidi Xiong","Shan Chen","Zhenting Qi","Himabindu Lakkaraju"],"authors_zh":"Zidi Xiong、Shan Chen、Zhenting Qi、Himabindu Lakkaraju","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reasoning-faithfulness","counterfactual-intervention","process-traces"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究推理过程忠实性、可监控性与干预评测的读者。","confidence":"high","one_line":["A counterfactual benchmark tests whether reasoning-model thinking drafts causally guide both subsequent steps and final answers.","以反事实干预测试推理模型的思维草稿是否真正影响后续推理与最终答案。"],"why":"It turns vague claims about visible reasoning into controlled, reusable tests of intra-draft and draft-to-answer dependence.","primary_link":"https://arxiv.org/abs/2505.13774","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/polaris-73/faithful-thinking-draft"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/polaris-73/faithful-thinking-draft"}],"link_count":4,"sections":9},{"id":"med-prm-guideline-verified-process-rewards-emnlp-2025","title":"Med-PRM: Medical Reasoning Models with Stepwise, Guideline-verified Process Rewards","year":2025,"venue":"EMNLP 2025","authors":["Jaehoon Yun","Jiwoong Sohn","Jungwoo Park et al."],"authors_zh":"Yun et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Med-PRM: Medical Reasoning Models with Stepwise, Guideline-verified Process Rewards exposes process or trace supervision data.","提出 Med-PRM，通过检索临床指南和文献逐步核验医学推理，并将证据驱动的过程评分用于诊疗问答。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://aclanthology.org/2025.emnlp-main.837/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/eth-medical-ai-lab/Med-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/dmis-lab/llama-3.1-medprm-reward-training-set"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/dmis-lab/llama-3.1-medprm-reward-v1.0"},{"key":"project","label":["Project","项目主页"],"url":"https://med-prm.github.io/"}],"link_count":5,"sections":9},{"id":"med-refl-medical-reflection-2025","title":"Med-REFL: Medical Reasoning Enhancement via Self-Corrected Fine-grained Reflection","year":2025,"venue":"arXiv","authors":["Hongxian Yang","Jiayu Qian","Segao Peng","Haoyu Zhang","Lu-An Huang","MC Tan","Shi-An Huang"],"authors_zh":"Hongxian Yang、Jiayu Qian、Segao Peng、Haoyu Zhang、Lu-An Huang、MC Tan、Shi-An Huang","tracks":["preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["trace_writing","reward_verifier_layer"],"domains":["medical-reasoning"],"tags":["medical","reflection","preference-pairs","dpo"],"status":"verified","priority":"可读","paper_type_zh":"医学推理偏好数据与自我反思方法论文","best_for_zh":"研究医学大模型、推理纠错或偏好优化的读者。","confidence":"medium","one_line":["Med-REFL creates self-corrected fine-grained reflection preference pairs for medical reasoning optimization.","Med-REFL 将细粒度自我反思转化为医学推理的偏好对，用于纠错式偏好优化。"],"why":"It supplies a preference-data object that targets error correction within medical reasoning rather than only final-answer selection.","primary_link":"https://arxiv.org/abs/2506.13793","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TianYin123/Med-REFL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HANI-LAB/Med-REFL-DPO"}],"link_count":4,"sections":9},{"id":"medreason-2025","title":"MedReason: Eliciting Factual Medical Reasoning Steps in LLMs via Knowledge Graphs","year":2025,"venue":"arXiv preprint","authors":["Juncheng Wu","Wenlong Deng","Xingxuan Li","Sheng Liu","Taomian Mi","Yifan Peng","Ziyang Xu","Yi Liu","Hyunjin Cho","Chang-In Choi","Yihan Cao","Hui Ren","Xiang Li","Xiaoxiao Li","Yuyin Zhou"],"authors_zh":"Juncheng Wu, Wenlong Deng, Xingxuan Li, Sheng Liu, Taomian Mi, Yifan Peng, Ziyang Xu, Yi Liu, Hyunjin Cho, Chang-In Choi, Yihan Cao, Hui Ren, Xiang Li, Xiaoxiao Li, Yuyin Zhou","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["medical-reasoning","clinical-question-answering","knowledge-graphs"],"tags":["medical-reasoning-data","knowledge-graph-grounding","chain-of-thought","answer-filtering","arxiv-2504.00993"],"status":"verified","priority":"必读","paper_type_zh":"医学推理数据集与知识图谱约束的合成流程","best_for_zh":"适合构建或审计医学 CoT 数据、知识图谱约束生成和答案级筛选流程的研究者。","confidence":"high","one_line":["MedReason maps medical QA entities through PrimeKG, uses GPT-4o to write grounded rationales, and releases 32,682 answer-filtered step-by-step SFT records.","MedReason 把医疗问答实体映射到 PrimeKG，由 GPT-4o 生成有知识路径约束的解释，并开放 32,682 条通过答案恢复筛选的逐步 SFT 记录。"],"why":"It makes factual guidance an explicit input to medical rationale generation and exposes the resulting records, rather than relying only on an unconstrained reasoning teacher.","primary_link":"https://arxiv.org/abs/2504.00993","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UCSC-VLAA/MedReason"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/UCSC-VLAA/MedReason"}],"link_count":5,"sections":9},{"id":"tokenizer-membership-inference-2025","title":"Membership Inference Attacks on Tokenizers of Large Language Models","year":2025,"venue":"USENIX Security 2026","authors":[],"authors_zh":"Meng Tong, Yuntao Du, Kejiang Chen, Weiming Zhang, Ninghui Li","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"必读","paper_type_zh":"预训练数据成员推断与 tokenizer 隐私审计论文","best_for_zh":"需要审计预训练语料来源、污染或 tokenizer 隐私泄漏的研究者。","confidence":"high","one_line":["tokenizer membership-inference datasets and defense code","以 tokenizer 词表为攻击面推断语料成员关系，提供更可控的 LLM 预训练数据泄漏审计路径。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://github.com/mengtong0110/Tokenizer-MIA","links":[],"link_count":2,"sections":9},{"id":"mimo-embodied-2025","title":"MiMo-Embodied: X-Embodied Foundation Model Technical Report","year":2025,"venue":"arXiv preprint","authors":["Xiaoshuai Hao","Lei Zhou","Zhijian Huang","Zhiwen Hou","Yingbo Tang","Lingfeng Zhang","Guang Li","Zheng Lu","Shuhuai Ren","Xianhui Meng","Yuchen Zhang","Jing Wu","Jinghui Lu","Chenxu Dang","Jiayi Guan","Jianhua Wu","Zhiyi Hou","Hanbing Li","Shumeng Xia","Mingliang Zhou","Yinan Zheng","Zihao Yue","Shuhao Gu","Hao Tian","Yuannan Shen","Jianwei Cui","Wen Zhang","Shaoqing Xu","Bing Wang","Haiyang Sun","Zeyu Zhu","Yuncheng Jiang","Zibin Guo","Chuhong Gong","Chaofan Zhang","Wenbo Ding","Kun Ma","Guang Chen","Rui Cai","Diyun Xiang","Heng Qu","Fuli Luo","Hangjun Ye","Long Chen"],"authors_zh":"Xiaoshuai Hao 等（Xiaomi MiMo）","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","process_supervision","construction_recipe","scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","state_action_level","scalar_reward"],"training_use":["sft","distillation","process_supervision","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["multimodal_reasoning","embodied_ai","autonomous_driving","affordance_prediction","task_planning","spatial_understanding","visual_grounding","trajectory_planning","status_prediction","environmental_perception","robotic_navigation","robotic_manipulation"],"tags":["mimo-embodied","xiaomi-mimo","embodied-ai","autonomous-driving","affordance","spatial-reasoning","trajectory-planning","chain-of-thought-sft","grpo","rule-based-reward","exact-match","iou-reward","point-in-mask","navsim","diffgrpo","open-loop-evaluation","sensor-privacy","train-eval-overlap-risk","sim-to-real-gap","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"跨具身多模态模型与自动驾驶训练技术报告","best_for_zh":"研究 embodied/driving 数据对象、规则奖励、轨迹规划、传感器隐私与 open-loop 安全审计的读者","confidence":"high","one_line":["MiMo-Embodied stages embodied, driving, generated-CoT, and deterministic-GRPO supervision, then separately studies NAVSIM IL+DiffGRPO, but discloses no training scale, mixtures, immutable splits, sensor ledger, rollouts, or planner artifacts.","MiMo-Embodied 以四阶段课程串联 embodied、driving、generated-CoT 与规则 GRPO，并另行研究 NAVSIM IL+DiffGRPO；但训练规模/比例、不可变 split、传感器账本、rollout 与 planner artifact 均未披露。"],"why":"It maps heterogeneous embodied data objects into one curriculum while showing why offline geometric rewards, open-loop planning, and open weights do not substitute for provenance, privacy, closed-loop safety, and environment verification.","primary_link":"https://arxiv.org/abs/2511.16518","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XiaomiMiMo/MiMo-Embodied"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/XiaomiMiMo/MiMo-Embodied-7B"}],"link_count":4,"sections":9},{"id":"mimo-vl-2025","title":"MiMo-VL Technical Report","year":2025,"venue":"arXiv preprint","authors":["Xiaomi LLM-Core Team"],"authors_zh":"Xiaomi LLM-Core Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","unknown"],"construction_layer":["trace_writing","frontier_pipeline","release_audit"],"domains":["multimodal","visual_reasoning"],"tags":["frontier-report","data-disclosure-ledger","multimodal-reasoning","long-cot","morl","rejection-sampling","open-checkpoint"],"status":"partial","priority":"可读","paper_type_zh":"前沿多模态模型技术报告与数据披露台账","best_for_zh":"审计长 CoT 预训练与混合在线 RL 披露边界的读者。","confidence":"medium","one_line":["MiMo-VL reports synthetic long-CoT regeneration with rejection sampling in later pre-training and mixed on-policy RL, but does not release the underlying traces, reward contract, or corpus audit.","MiMo-VL 报告 2.4T token 四阶段预训练、长 CoT 推理数据与 MORL，但未披露数据来源、奖励契约、rollout 或审计细节。"],"why":"It provides a useful disclosure ledger for a frontier multimodal reasoner because it separates released checkpoints and evaluation prompts from undisclosed data provenance, acceptance decisions, and mixed-RL feedback details.","primary_link":"https://arxiv.org/abs/2506.03569","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XiaomiMiMo/MiMo-VL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/XiaomiMiMo/MiMo-VL-7B-RL"}],"link_count":3,"sections":9},{"id":"mimo-reasoning-pretraining-posttraining-2025","title":"MiMo: Unlocking the Reasoning Potential of Language Model -- From Pretraining to Posttraining","year":2025,"venue":"arXiv preprint","authors":["LLM-Core Xiaomi"],"authors_zh":"小米 LLM-Core 团队","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","verifier_reward","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["sft","distillation","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","coding","reasoning","general_language","long_context"],"tags":["mimo","xiaomi","frontier-report","pretraining-data","distillation","sft","rlvr","rule-verifier","code-reward","dynamic-sampling","rollout-engine","partial-release","disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型报告与数据披露账本","best_for_zh":"审计规则验证 RL、代码测试奖励、开放权重与未公开数据/审计链之间的边界","confidence":"high","one_line":["MiMo reports 25T-token reasoning-oriented pretraining, 500K then 6M distilled SFT instances, and 130K rule-verifiable RL problems, while releasing weights and inference code but not the training data, tests, verifiers, rollouts, or complete training stack.","MiMo 报告 13 万条可规则验证的数学/代码 RL 题、测试难度奖励、重采样与公开 checkpoints；任务记录、验证器细节、奖励标定和审计资产仍未公开。"],"why":"It shows why a frontier data ledger must separate pretraining composition, SFT distillation, RL tasks and rewards, actual open artifacts, and benchmark evidence rather than treating open weights as a reusable data recipe.","primary_link":"https://arxiv.org/abs/2505.07608","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XiaomiMiMo/vllm/tree/feat_mimo_mtp_stable_073"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/XiaomiMiMo/MiMo-7B-RL"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/XiaomiMiMo/MiMo"}],"link_count":6,"sections":9},{"id":"minimax-m1-2025","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","year":2025,"venue":"arXiv preprint","authors":["MiniMax"],"authors_zh":"MiniMax","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","scaling_study","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","reward_modeling","rlvr","agent_training","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["mathematics","coding","software_engineering","general"],"tags":["minimax-m1","frontier-report","data-disclosure-ledger","open-weights","long-cot","cispo","rlvr","execution-feedback","generative-reward-model","long-context","test-time-compute"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型技术报告与数据披露账本","best_for_zh":"审计长上下文推理模型的 RL、沙盒任务与开放权重边界的读者","confidence":"high","one_line":["MiniMax-M1 releases 40K and 80K model weights and reports separate continual-pretraining, Long-CoT SFT, and mixed-feedback RL recipes, but not the training rows, reward traces, sandboxes, or training code needed for replay.","MiniMax-M1 披露混合注意力推理模型、面向含沙盒软件工程问题的大规模 RL 与 CISPO，但未公开训练数据、奖励或 rollout 审计工件。"],"why":"It is a useful frontier-report ledger for comparing what is disclosed about source mixtures, filtering, verifiers, execution environments, curricula, weights, and evaluation against what remains unavailable.","primary_link":"https://arxiv.org/abs/2506.13585","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MiniMax-AI/MiniMax-M1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/MiniMaxAI/minimax-m1"},{"key":"project","label":["Project","项目主页"],"url":"https://www.minimax.io/news/minimaxm1"}],"link_count":6,"sections":9},{"id":"mir-bench-many-shot-in-context-reasoning-2025","title":"MIR-Bench: Can Your LLM Recognize Complicated Patterns via Many-Shot In-Context Reasoning?","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks","authors":["Kai Yan","Zhan Ling","Kang Liu","Yifan Yang","Ting-Han Fan","Lingfeng Shen","Zhengyin Du","Jiecao Chen"],"authors_zh":"Kai Yan、Zhan Ling、Kang Liu、Yifan Yang、Ting-Han Fan、Lingfeng Shen、Zhengyin Du、Jiecao Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["in-context-learning","reasoning"],"tags":["in-context-learning","pattern-reasoning","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"多示例上下文模式推理基准","best_for_zh":"研究长上下文、模式归纳与 many-shot 推理的研究者。","confidence":"high","one_line":["MIR-Bench evaluates whether models can infer complex patterns from many in-context demonstrations.","MIR-Bench 以多示例上下文中的复杂模式归纳任务，评估模型从大量演示中提炼规则的能力。"],"why":"MIR-Bench evaluates whether models can infer complex patterns from many in-context demonstrations.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/796076672b00f54fb01d05a2e5fde363-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/kaiyan289/MIR-Bench"}],"link_count":2,"sections":9},{"id":"miromind-m1-a-multi-stage-framework-for-reasoning","title":"MiroMind-M1: A Multi-Stage Framework for Reasoning","year":2025,"venue":"arXiv","authors":["Xingxuan Li","Yao Xiao","Dianwen Ng","Hai Ye","Yue Deng","Xiang Lin","Bin Wang","Zhanfeng Mo","Chong Zhang","Yueyi Zhang","Zonglin Yang","Ruilin Li","Lei Lei","Shihao Xu","Han Zhao","Weiling Chen","Feng Ji","Lidong Bing"],"authors_zh":"Xingxuan Li, Yao Xiao, Dianwen Ng, Hai Ye, Yue Deng, Xiang Lin, Bin Wang, Zhanfeng Mo, Chong Zhang, Yueyi Zhang, Zonglin Yang, Ruilin Li, Lei Lei, Shihao Xu, Han Zhao, Weiling Chen, Feng Ji, Lidong Bing","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** MiroMind-M1 combines 719K long-CoT SFT records, 62K verifiable RL prompts, and length-progressive optimization in a fully open stack.","MiroMind-M1 将 719K 长 CoT SFT、62K 可验证 RL 问题和长度渐进优化组成完整开放训练栈。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2507.14683","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MiroMindAsia/MiroMind-M1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/miromind-ai/MiroMind-M1-SFT-719K"},{"key":"project","label":["Project","项目主页"],"url":"https://miromind.ai/"}],"link_count":4,"sections":9},{"id":"mlalgo-bench-machine-learning-algorithms-2025","title":"MLAlgo-Bench: Can Machines Implement Machine Learning Algorithms?","year":2025,"venue":"Findings of EMNLP 2025","authors":["Yunfei Wang","Yeqin Zhang","Yuyang Wu","Liang Lu","Phi Le Nguyen","Xiaoliang Wang","Cam-Tu Nguyen"],"authors_zh":"Yunfei Wang, Yeqin Zhang, Yuyang Wu, Liang Lu, Phi Le Nguyen, Xiaoliang Wang, Cam-Tu Nguyen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","machine-learning","programmatic-evaluation"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"机器学习算法实现与运行验证基准","best_for_zh":"需要评估机器学习算法代码生成、性能约束或可执行实验复现的研究者。","confidence":"high","one_line":["MLAlgo-Bench tests machine-learning algorithm implementation with sandboxed execution, performance references, and runtime-aware validation.","MLAlgo-Bench 以算法实现和真实应用两类任务检验机器学习代码，使用沙箱执行、性能参照和时间开销自动验证。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://aclanthology.org/2025.findings-emnlp.772/","links":[{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/drive/folders/1qUWlvXX-VCZL9p3vn70a1svfOiZjsYCk"}],"link_count":2,"sections":9},{"id":"mlgym-2025","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","year":2025,"venue":"COLM 2025","authors":["Deepak Nathani","Lovish Madaan","Nicholas Roberts","Nikolay Bashlykov","Ajay Menon","Vincent Moens","Mikhail Plekhanov","Amar Budhiraja","Despoina Magka","Vladislav Vorotilov","Gaurav Chaurasia","Dieuwke Hupkes","Ricardo Silveira Cabral","Tatiana Shavrina","Jakob Nicolaus Foerster","Yoram Bachrach","William Yang Wang","Roberta Raileanu"],"authors_zh":"Deepak Nathani、Lovish Madaan、Nicholas Roberts、Nikolay Bashlykov、Ajay Menon、Vincent Moens、Mikhail Plekhanov、Amar Budhiraja、Despoina Magka、Vladislav Vorotilov、Gaurav Chaurasia、Dieuwke Hupkes、Ricardo Silveira Cabral、Tatiana Shavrina、Jakob Nicolaus Foerster、Yoram Bachrach、William Yang Wang、Roberta Raileanu","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","machine_learning_research","scientific_agents","code_agents","tool_use","benchmark_evaluation"],"tags":["environment-agent-trajectory-data","ai-research-agents","agent-environment","agent-trajectories","shell-tools","flexible-artifact-evaluation","programmatic-evaluator","test-feedback-adaptation","failed-trajectory-retention","replay-partial","release-version-drift","mixed-license"],"status":"partial","priority":"必读","paper_type_zh":"AI 研究智能体环境、评测基准与轨迹发布论文","best_for_zh":"研究环境交互轨迹、科学智能体评测、程序化 evaluator、失败样本保留、replay 与数据发布审计的读者","confidence":"high","one_line":["MLGym releases 13 containerized ML-research tasks and a current 676-run action/observation corpus with task-specific evaluator scores and replay code, but repeated validation exposes test feedback, native Gym reward remains zero, and the release is untagged and license-mixed.","MLGym 将 13 个容器化 ML 研究任务表示为 thought/action/observation/state 与 evaluator score 轨迹；论文分析 624 条运行，而当前仓库为 676 对轨迹/结果文件，但无限 test-set validate、恒为 0 的 Gym reward、可变 latest 镜像与混合许可限制其复用。"],"why":"It connects task/data configuration, container state, one-command agent actions, observations, intermediate evaluator feedback, flexible research artifacts, failures, and replay into one auditable episode object, while showing why a Gym interface and public trajectories do not by themselves establish a clean RL reward, held-out evaluation, deterministic replay, or training-safe release.","primary_link":"https://openreview.net/pdf/75f6e6aa5276a0b93fd3859ec7b41c92ee79cea8.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/MLGym"},{"key":"data","label":["Data","数据"],"url":"https://github.com/facebookresearch/MLGym/tree/9d40c1b5035202018cd7091fb4e83a9c68b377c0/trajectories/mlgym_bench_v0"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/mlgym/coco-captioning"}],"link_count":9,"sections":9},{"id":"mm-browsecomp-2025","title":"MM-BrowseComp: A Comprehensive Benchmark for Multimodal Browsing Agents","year":2025,"venue":"arXiv preprint","authors":["Shilong Li","Xingyuan Bu","Wenjie Wang","Jiaheng Liu","Jun Dong","Haoyang He","Hao Lu","Haozhe Zhang","Chenchen Jing","Zhen Li","Chuanhao Li","Jiayi Tian","Chenchen Zhang","Tianhao Peng","Yancheng He","Jihao Gu","Yuanxing Zhang","Jian Yang","Ge Zhang","Wenhao Huang","Wangchunshu Zhou","Zhaoxiang Zhang","Ruizhe Ding","Shilei Wen"],"authors_zh":"Shilong Li 等（ByteDance、Nanjing University、M-A-P、CASIA 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit","reward_verifier_layer"],"domains":["multimodal-browsing","live-hidden-contamination-audit"],"tags":["benchmark","live_hidden_contamination_audit","multimodal-browsing"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["MM-BrowseComp exposes multimodal browsing questions involving image/video evidence as an auditable evaluation surface.","MM-BrowseComp 把需要图像与视频证据的多模态浏览问答做成可审计的评测面。"],"why":"Promising successor to BrowseComp but exact source/version needs source audit.","primary_link":"https://arxiv.org/abs/2508.13186","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MMBrowseComp/MM-BrowseComp"}],"link_count":3,"sections":9},{"id":"mm-opera-open-ended-association-2025","title":"MM-OPERA: Benchmarking Open-ended Association Reasoning for Large Vision-Language Models","year":2025,"venue":"NeurIPS 2025","authors":["Zimeng Huang","Jinxin Ke","Xiaoxuan Fan","Yufeng Yang","Yang Liu","Liu Zhonghan","Zedi Wang","Junteng Dai","Haoyi Jiang","Yuyu Zhou","Keze Wang","Ziliang Chen"],"authors_zh":"Zimeng Huang、Jinxin Ke、Xiaoxuan Fan、Yufeng Yang、Yang Liu、Liu Zhonghan、Zedi Wang、Junteng Dai、Haoyi Jiang、Yuyu Zhou、Keze Wang、Ziliang Chen","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["MM-OPERA evaluates open-ended multimodal association reasoning with process-aware judging of divergent and convergent associations.","MM-OPERA 以开放式图文联想任务和过程感知裁判，评估视觉语言模型的发散与收敛关联推理。"],"why":"MM-OPERA evaluates open-ended multimodal association reasoning with process-aware judging of divergent and convergent associations.","primary_link":"https://arxiv.org/abs/2510.26937","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MM-OPERA-Bench/MM-OPERA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/titic/MM-OPERA"}],"link_count":3,"sections":9},{"id":"mm-prm-2025","title":"MM-PRM: Enhancing Multimodal Mathematical Reasoning with Scalable Step-Level Supervision","year":2025,"venue":"arXiv","authors":["Lingxiao Du","Fanqing Meng","Zongkai Liu","Zhixiang Zhou","Ping Luo","Qiaosheng Zhang","Wenqi Shao"],"authors_zh":"Lingxiao Du、Fanqing Meng、Zongkai Liu、Zhixiang Zhou、Ping Luo、Qiaosheng Zhang、Wenqi Shao","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["process_supervision","verifier_reward","construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","test_time_compute"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["multimodal","math"],"tags":["multimodal-math","process-reward-model","mcts","rollout-value","step-level-supervision","soft-labels","best-of-n","open-code","partial-release"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹、搜索或测试时计算研究","best_for_zh":"需要审计 rollout、选择器、预算、数据谱系与复现边界的读者","confidence":"high","one_line":["An OmegaPRM-style multimodal pipeline that turns answer-judged rollout success rates into soft step labels and trains an 8B PRM, while releasing the seed data and checkpoint but not the reported annotation trees.","该条目将推理轨迹、搜索选择或测试时计算作为可审计的研究对象；未披露字段已明确标为 unknown。"],"why":"It makes the chain from multimodal rollouts to step rewards explicit and shows why Monte Carlo value, final-answer judgment, learned PRM score, and Best-of-N selection must be audited as separate contracts.","primary_link":"https://arxiv.org/abs/2505.13427","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ModalMinds/MM-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Cierra0506/MM-K12"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Cierra0506/MM-PRM"}],"link_count":5,"sections":9},{"id":"mm-rlhf-the-next-step-forward-in-multimodal-llm-alignment-2025","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","year":2025,"venue":"ICML 2025","authors":["Yi-Fan Zhang","Tao Yu","Haochen Tian","Chaoyou Fu","Peiyan Li","Jianshu Zeng","Wulin Xie","Yang Shi","Huanyu Zhang","Junkang Wu","Xue Wang","Yibo Hu","Bin Wen","Fan Yang","Zhang Zhang","Tingting Gao","Di Zhang","Liang Wang","Rong Jin","Tieniu Tan"],"authors_zh":"Yi-Fan Zhang、Tao Yu、Haochen Tian、Chaoyou Fu 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"多模态人工偏好数据、奖励模型与偏好优化论文","best_for_zh":"研究视觉语言模型对齐、多模态奖励建模或偏好学习的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","MM-RLHF 发布大规模人工标注的多模态偏好比较对，并配套奖励模型与偏好优化方法。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2502.10391","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yfzhang114/MM-RLHF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yifanzhang114/MM-RLHF"},{"key":"project","label":["Project","项目主页"],"url":"https://mm-rlhf.github.io/"}],"link_count":4,"sections":9},{"id":"mmat-1m-multimodal-agent-tuning-2025","title":"MMAT-1M: A Large Reasoning Dataset for Multimodal Agent Tuning","year":2025,"venue":"ICCV 2025","authors":["Tianhong Gao","Yannian Fu","Weiqun Wu","Haixiao Yue","Shanshan Liu","Gang Zhang"],"authors_zh":"Tianhong Gao, Yannian Fu, Weiqun Wu, Haixiao Yue, Shanshan Liu, Gang Zhang","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","tool-use","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"百万级多模态推理、反思与工具使用智能体数据集论文","best_for_zh":"需要同时训练多模态链式推理、自我反思和动态工具使用能力的研究者。","confidence":"high","one_line":["MMAT-1M releases million-scale multimodal Rationale-and-Reflection dialogues with dynamic API and retrieval interactions.","MMAT-1M 将多模态问答扩展为包含推理、反思、API 调用与 RAG 的百万级智能体训练对话。"],"why":"It makes both multi-turn and compressed process-supervision formats publicly available at substantially larger scale than earlier multimodal agent sets.","primary_link":"https://openaccess.thecvf.com/content/ICCV2025/html/Gao_MMAT-1M_A_Large_Reasoning_Dataset_for_Multimodal_Agent_Tuning_ICCV_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/VIS-MPU-Agent/MMAT-1M"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/VIS-MPU-Agent/MMAT-1M"}],"link_count":4,"sections":9},{"id":"mmsearch-plus-2025","title":"MMSearch-Plus: Benchmarking Provenance-Aware Search for Multimodal Browsing Agents","year":2025,"venue":"ICLR 2026 Poster","authors":["Xijia Tao","Yihua Teng","Xinxing Su","Xinyu Fu","Jihao Wu","Chaofan Tao","Ziru Liu","Haoli Bai","Rui Liu","Lingpeng Kong"],"authors_zh":"Xijia Tao, Yihua Teng, Xinxing Su, Xinyu Fu, Jihao Wu, Chaofan Tao, Ziru Liu, Haoli Bai, Rui Liu, Lingpeng Kong","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","release_audit"],"domains":["multimodal_reasoning","web_browsing","agent_trajectories","environment_interaction","visual_question_answering"],"tags":["multimodal-browsing","web-search-agent","visual-question-answering","spatial-temporal-extrapolation","set-of-mark","dynamic-web-environment","llm-as-judge","answer-level-evaluation","provenance-aware-search","release-gap","replay-risk","temporal-drift","contamination-risk"],"status":"partial","priority":"可读","paper_type_zh":"多模态网页搜索智能体基准与数据发布","best_for_zh":"研究多模态浏览智能体、动态搜索环境、答案级judge、轨迹发布边界与污染审计的读者","confidence":"high","one_line":["MMSearch-Plus releases 311 image-grounded QA rows and evaluates agents over SerpAPI/Gemini search with answer-level GPT-4o judgment, but does not release the agent framework, SoM boxes, complete trajectories, or replay snapshots.","MMSearch-Plus发布单一train split中的311条图像问答与来源引用，并以SerpAPI/Gemini动态搜索和GPT-4o答案judge评测智能体；完整轨迹、框架、evaluator与SoM均未发布。"],"why":"It makes weak visual cues, region-level crop/search decisions, noisy live-web retrieval, and provenance checking part of the agent evaluation contract while showing why a public question/image table alone is insufficient for trajectory reuse, verifier audit, or deterministic comparison.","primary_link":"https://arxiv.org/abs/2508.21475","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mmsearch-plus/MMSearch-Plus"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Cie1/MMSearch-Plus"},{"key":"project","label":["Project","项目主页"],"url":"https://mmsearch-plus.github.io/"}],"link_count":11,"sections":9},{"id":"mobilerl-2025","title":"MobileRL: Online Agentic Reinforcement Learning for Mobile GUI Agents","year":2025,"venue":"arXiv preprint","authors":["Yifan Xu","Xiao Liu","Xinghan Liu","Jiaqi Fu","Hanchen Zhang","Bohao Jing","Shudan Zhang","Yuting Wang","Wenyi Zhao","Yuxiao Dong"],"authors_zh":"Yifan Xu, Xiao Liu, Xinghan Liu, Jiaqi Fu, Hanchen Zhang, Bohao Jing, Shudan Zhang, Yuting Wang, Wenyi Zhao, Yuxiao Dong","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","environment_interaction","mobile_gui","android","online_rl"],"tags":["environment-agent-trajectory-data","mobile-gui-agent","android","online-agentic-rl","rlvr","grpo","positive-replay","failure-curriculum","learned-reward-model","evaluation-code-only"],"status":"partial","priority":"可读","paper_type_zh":"移动 GUI 智能体在线强化学习 recipe 与混合 verifier 研究","best_for_zh":"研究移动端 agentic RL、轨迹 replay、失败筛选、终局 reward 与可复现 Android 环境的读者","confidence":"high","one_line":["MobileRL combines 97.9k action-only SFT steps, 23.6k reasoning-SFT steps, and online Android rollouts over 3,103 training tasks with mixed terminal verification, length shaping, positive replay, and failure filtering.","MobileRL 用 97.9k 步 action-only SFT、23.6k 步 reasoning SFT 和 3,103 个在线 Android 训练任务构建移动端 RL pipeline，并以混合终局 verifier、路径长度 shaping、正向 replay 和失败课程控制轨迹分布，但训练代码、数据和 replay 尚未发布。"],"why":"It turns the trajectory distribution itself into a dynamic post-training object: scarce difficult successes are replayed, low-advantage failures are pruned, repeatedly unsolved tasks are removed, and two different verifier regimes determine which mobile behaviors receive policy gradients.","primary_link":"https://arxiv.org/abs/2509.18119","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/MobileRL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/xuyifan/MobileRL-9B"}],"link_count":5,"sections":9},{"id":"mobileworldbench-2025","title":"MobileWorldBench: Towards Semantic World Modeling For Mobile Agents","year":2025,"venue":"arXiv preprint","authors":["Shufan Li","Konstantinos Kallidromitis","Akash Gokul","Yusuke Kato","Kazuki Kozuka","Aditya Grover"],"authors_zh":"Shufan Li, Konstantinos Kallidromitis, Akash Gokul, Yusuke Kato, Kazuki Kozuka, Aditya Grover","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","benchmark","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","answer_level","scalar_reward"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","mobile_gui","semantic_world_models","state_transition_prediction"],"tags":["mobile-gui","semantic-world-model","state-transition","next-state-generation","next-state-qa","vlm-judge","synthetic-annotation","human-filtering","agent-trajectories","offline-benchmark","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"移动GUI单步语义世界模型数据集与离线混合反馈基准","best_for_zh":"研究移动智能体SFT数据、next-state预测、VLM judge与离线评测审计的读者","confidence":"medium","one_line":["MobileWorldBench evaluates one-step Android future-state semantics with GPT-judged descriptions and yes/no QA, while MobileWorld supplies roughly 1.4M model-annotated transition items for SFT.","MobileWorldBench把移动智能体行为压缩为离线单步state/action/next-state预测，并以GPT-4o rubric和Yes/No准确率混合评测；MobileWorld另供SFT数据，但split、图像完整性、judge版本、许可与隐私仍有缺口。"],"why":"It isolates the predictive component of mobile-agent reasoning into auditable state/action/next-state records, but also shows why source trajectories do not imply episode supervision and why judge versions, split hashes, replay support, licenses, and annotation lineage matter before reuse.","primary_link":"https://arxiv.org/abs/2512.14014","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/jacklishufan/MobileWorld"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jacklishufan/MobileWorld"}],"link_count":7,"sections":9},{"id":"more-data-or-better-data-2025","title":"More Data or Better Data? A Critical Analysis of Data Selection and Synthesis for Mathematical Reasoning","year":2025,"venue":"EMNLP 2025 Industry Track","authors":["Yike Zhao","Simin Guo","Ziqing Yang","Shifan Han","Dahua Lin","Fei Tan"],"authors_zh":"Yike Zhao、Simin Guo、Ziqing Yang、Shifan Han、Dahua Lin、Fei Tan","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","scaling_report"],"domains":["mathematical_reasoning"],"tags":["math-data-selection","data-synthesis","controlled-mixtures","negative-results","release-gaps"],"status":"partial","priority":"必读","paper_type_zh":"数学推理数据选择与合成比较研究","best_for_zh":"关注数学推理数据配方、受控比较和开放发布审计的研究者","confidence":"medium","one_line":["Compares math-data recipes under a fixed 80/20 scaffold, finding benefits from restructuring and strong-model distillation but releasing none of the constructed corpora.","在固定 80% 基线与 20% 候选混合中比较数学数据配方，教育化重构和 QwQ 蒸馏仅改善部分数学指标，且存在跨任务回退、代码与构造语料未公开。"],"why":"Separates practical data-recipe effects from simple scale while exposing missing controls, lineage and release artifacts.","primary_link":"https://aclanthology.org/2025.emnlp-industry.43/","links":[],"link_count":4,"sections":9},{"id":"mpbench-multimodal-process-errors-2025","title":"MPBench: A Comprehensive Multimodal Reasoning Benchmark for Process Errors Identification","year":2025,"venue":"Findings of ACL 2025","authors":["Zhaopan Xu","Pengfei Zhou","Jiaxin Ai","Wangbo Zhao","Kai Wang","Xiaojiang Peng","Wenqi Shao","Hongxun Yao","Kaipeng Zhang"],"authors_zh":"Zhaopan Xu, Pengfei Zhou, Jiaxin Ai, Wangbo Zhao, Kai Wang, Xiaojiang Peng, Wenqi Shao, Hongxun Yao, Kaipeng Zhang","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","process-reward-modeling","evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要在视觉、文本与中间推理联合条件下评测过程奖励模型的研究者。","confidence":"high","one_line":["MPBench evaluates multimodal PRMs through step correctness, answer aggregation, and reasoning-process search.","MPBench 以首错步骤和错误类型标注评测多模态过程奖励模型的步骤判断、答案聚合与过程搜索。"],"why":"The release provides a reusable process-supervision or process-evaluation surface.","primary_link":"https://aclanthology.org/2025.findings-acl.1112/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MPBench/MPBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xuzhaopan/MPBench"},{"key":"project","label":["Project","项目主页"],"url":"https://mpbench.github.io/"}],"link_count":5,"sections":9},{"id":"mr-gsm8k-meta-reasoning-benchmark-2025","title":"MR-GSM8K: A Meta-Reasoning Benchmark for Large Language Model Evaluation","year":2025,"venue":"ICLR 2025","authors":["Zhongshen Zeng","Pengguang Chen","Shu Liu","Haiyun Jiang","Jiaya Jia"],"authors_zh":"Zhongshen Zeng、Pengguang Chen、Shu Liu、Haiyun Jiang、Jiaya Jia","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","meta-reasoning","process-evaluation"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测模型能否审阅数学推理、定位首错，或构建过程判别器基准的研究者。","confidence":"high","one_line":["MR-GSM8K turns GSM8K solutions into meta-reasoning tasks that require models to score a trace and pinpoint its first error, exposing process-evaluation gaps hidden by answer accuracy.","MR-GSM8K 将 GSM8K 解答转为评判推理与定位首错的元推理任务，揭示仅看最终答案会掩盖的过程评估差距。"],"why":"It makes explicit error localization and solution assessment a first-class evaluation target with a reusable structured test schema.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/fc0b0e6ac2da44d5839b13f90625b357-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JIA-Lab-research/MR-GSM8K"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Randolphzeng/Mr-GSM8K"}],"link_count":5,"sections":9},{"id":"mr3-multilingual-rubric-agnostic-reward-reasoning-models-2025","title":"mR3: Multilingual Rubric-Agnostic Reward Reasoning Models","year":2025,"venue":"ICLR 2026","authors":["David Anugraha","Shou-Yi Hung","Zilu Tang","Annie En-Shiun Lee","Derry Tanti Wijaya","Genta Indra Winata"],"authors_zh":"David Anugraha、Shou-Yi Hung、Zilu Tang、Annie En-Shiun Lee、Derry Tanti Wijaya 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","mR3 将量规式奖励推理扩展到 72 种语言，并以多语种课程数据训练统一评审模型。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2510.01146","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rubricreward/mr3"}],"link_count":2,"sections":9},{"id":"mt-rewardtree-2025","title":"MT-RewardTree: A Comprehensive Framework for Advancing LLM-Based Machine Translation via Reward Modeling","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Zhaopeng Feng","Jiahan Ren","Jiayuan Su","Jiamei Zheng","Hongwei Wang","Zuozhu Liu"],"authors_zh":"Zhaopeng Feng、Jiahan Ren、Jiayuan Su、Jiamei Zheng、Hongwei Wang、Zuozhu Liu","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","benchmark","verifier_reward","process_supervision","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","pairwise_preference","scalar_reward","process_reward","trajectory_value"],"training_use":["preference_learning","reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer"],"domains":["machine_translation","multilingual_nlp","english","german","chinese","russian"],"tags":["mt-rewardtree","machine-translation","approximate-mcts","token-level-preference","sibling-preference","rollout-values","cometkiwi","process-reward-model","dpo","kto","reward-benchmark","test-time-alignment","raw-tree-missing","model-judge-risk"],"status":"partial","priority":"可读","paper_type_zh":"机器翻译搜索偏好数据、过程奖励模型与评测基准","best_for_zh":"研究 token 级搜索数据、PRM、偏好学习、测试时对齐及 learned-metric 审计的读者","confidence":"medium","one_line":["MT-RewardTree turns top-2 translation-token branches and three COMETKiwi-scored rollouts per node into 8,652 prefixed training pairs and a 1,200-pair benchmark, while releasing only selected pairs rather than the raw search trees.","MT-RewardTree 以 top-2 同前缀 token 分支和每节点三次 COMETKiwi 评分 rollout 构造 8,652 组偏好对及 1,200 组 benchmark，但公开数据只保留筛选后的 chosen/rejected，而不含原始搜索树。"],"why":"It exposes a concrete recipe for converting token-level search and trajectory values into PRM preference supervision and reward-guided decoding, while showing that metric dependence, hidden rejected rollouts, split ambiguity, and unresolved source/model licensing can bound reuse.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1007/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sabijun/MT_RewardTree"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/sabijun/mt-rewardtree-dataset"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/sabijun"},{"key":"project","label":["Project","项目主页"],"url":"https://sabijun.github.io/MT_RewardTreePage/"}],"link_count":10,"sections":9},{"id":"mulberry-reasoning-reflection-2025","title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","year":2025,"venue":"NeurIPS 2025","authors":["Huanjin Yao","Jiaxing Huang","Wenhao Wu","Jingyi Zhang","Yibo Wang","Shunyu Liu","Yingjie Wang","Yuxin Song","Haocheng Feng","Li Shen","Dacheng Tao"],"authors_zh":"Huanjin Yao, Jiaxing Huang, Wenhao Wu, Jingyi Zhang, Yibo Wang, Shunyu Liu, Yingjie Wang, Yuxin Song, Haocheng Feng, Li Shen, Dacheng Tao","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","search_substrate","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","visual-question-answering","mathematics","charts","science"],"tags":["multimodal-rationale-data","reflection-data","collective-tree-search","teacher-distillation","visual-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"多模态推理与反思数据集及集体搜索配方","best_for_zh":"适合构建或审计图像推理、反思 SFT 数据的读者。","confidence":"high","one_line":["Mulberry-260K uses four-model tree search to release image-grounded reasoning and reflection demonstrations for multimodal SFT.","Mulberry-260K 用四模型树搜索发布图像推理与反思示范，供多模态 SFT 使用。"],"why":"It makes positive paths, sampled error-to-correction transitions, search budgets, and an SFT-ready release visible in one multimodal recipe.","primary_link":"https://arxiv.org/abs/2412.18319","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HJYao00/Mulberry"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuanjinYao/Mulberry-SFT"}],"link_count":5,"sections":9},{"id":"matpo-multi-agent-tool-policy-2025","title":"Multi-Agent Tool-Integrated Policy Optimization","year":2025,"venue":"AAAI 2026 TrustAgent Workshop Poster","authors":["Zhanfeng Mo","Xingxuan Li","Yuntao Chen","Lidong Bing"],"authors_zh":"Zhanfeng Mo、Xingxuan Li、Yuntao Chen、Lidong Bing","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multi-agent-rl","tool-use","process-rewards"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究多智能体工具调用强化学习、信用分配与轨迹监督的读者。","confidence":"high","one_line":["MATPO trains planner and worker roles within one LLM by assigning joint tool-use rollout rewards across nested agent trajectories.","MATPO 在单一 LLM 内训练规划者与工作者角色，并对嵌套工具使用轨迹进行联合奖励归因。"],"why":"It releases large-scale scored multi-agent and single-agent tool-use rollouts together with a practical credit-assignment objective.","primary_link":"https://arxiv.org/abs/2510.04678","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mzf666/MATPO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/veggiebird/MATPO-rollout"}],"link_count":5,"sections":9},{"id":"multi-agent-verification-2025","title":"Multi-Agent Verification: Scaling Test-Time Compute with Multiple Verifiers","year":2025,"venue":"ICLR 2025 MCDC Workshop","authors":["Shalev Lifshitz","Sheila A. McIlraith","Yilun Du"],"authors_zh":"Shalev Lifshitz、Sheila A. McIlraith、Yilun Du（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["verifier_reward","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","verifier-ensemble","best-of-n"],"status":"verified","priority":"可读","paper_type_zh":"多验证器测试时扩展论文","best_for_zh":"研究推理预算下验证器集成的读者。","confidence":"high","one_line":["MAV scales test-time verification by aggregating multiple aspect-specific verifiers over best-of-n candidates.","MAV 通过聚合多个方面验证器来扩展 best-of-n 候选的测试时验证。"],"why":"It separates gains from more verifier judgments from gains from more generated candidates.","primary_link":"https://arxiv.org/abs/2502.20379","links":[],"link_count":2,"sections":9},{"id":"multimodal-agent-tuning-2025","title":"Multi-modal Agent Tuning: Building a VLM-Driven Agent for Efficient Tool Usage","year":2025,"venue":"ICLR 2025 Spotlight","authors":["Zhi Gao","Bofei Zhang","Pengxiang Li","Xiaojian Ma","Tao Yuan","Yue Fan","Yuwei Wu","Yunde Jia","Song-Chun Zhu","Qing Li"],"authors_zh":"Zhi Gao、Bofei Zhang、Pengxiang Li、Xiaojian Ma、Tao Yuan、Yue Fan、Yuwei Wu、Yunde Jia、Song-Chun Zhu、Qing Li","tracks":["data_construction_open_release_recipes","process_trace_supervision_data"],"source_role":["construction_recipe","data_release","agent_environment"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["multimodal_agents","tool_use","visual_reasoning","code_generation"],"tags":["multimodal-agent-tuning","mm-traj","t3-agent","multimodal-agents","tool-use","react","gpt-4o-mini","synthetic-trajectories","executable-code","llm-as-judge","dual-verification","sft","source-rights","license-conflict","schema-drift","rejected-data-gap","release-completeness"],"status":"partial","priority":"必读","paper_type_zh":"多模态 agent 轨迹构造 recipe、数据发布与模型研究","best_for_zh":"研究多模态工具使用 SFT、合成 ReAct 轨迹、LLM-as-judge 筛选或开放数据审计的读者","confidence":"high","one_line":["Multi-modal Agent Tuning synthesizes queries and files, executes GPT-4o-mini-controlled ReAct trajectories, applies query-file and trajectory judges, and fine-tunes VLM controllers on thought-and-code supervision.","Multi-modal Agent Tuning 用 GPT-4o mini 生成查询、文件与 ReAct 轨迹并经可执行代码门槛和双重同源 judge 筛选，公开 21,168 条 MM-Traj 记录；其构造 recipe 可供研究，但许可冲突、schema 失败、缺少拒绝账本与独立正确性核验阻断直接训练复用。"],"why":"It documents an unusually inspectable multimodal agent-data pipeline while showing why executable traces and two judges are insufficient without independent correctness checks, source-rights inheritance, stable row/file lineage, rejected examples, and reproducible release manifests.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/238747e153a84f50b43fd50fa8504f33-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mat-agent/MAT-Agent"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PengxiangLi/MAT"},{"key":"project","label":["Project","项目主页"],"url":"https://mat-agent.github.io/"}],"link_count":11,"sections":9},{"id":"multi-swe-bench-multilingual-issue-resolution-2025","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Daoguang Zan","Zhirong Huang","Wei Liu","Hanwu Chen","Linhao Zhang","Shulin Xin","Lu Chen","Qi Liu","Xiaojian Zhong","Aoyan Li","Siyao Liu","Yongsheng Xiao","Liangqiang Chen","Yuyu Zhang","Jing Su","Tianyu Liu","Rui Long","Kai Shen","Liang Xiang"],"authors_zh":"Daoguang Zan, Zhirong Huang, Wei Liu, Hanwu Chen, Linhao Zhang, Shulin Xin, Lu Chen, Qi Liu, Xiaojian Zhong, Aoyan Li, Siyao Liu, Yongsheng Xiao, Liangqiang Chen, Yuyu Zhang, Jing Su, Tianyu Liu, Rui Long, Kai Shen, Liang Xiang","tracks":["programmatically_verifiable_outcome_data","environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["software-engineering","multilingual","docker","issue-resolution","2025"],"status":"verified","priority":"必读","paper_type_zh":"多语言仓库级 issue 修复 benchmark 与可执行评测数据","best_for_zh":"需要多语言代码 agent 评测、Docker 验证或 issue 修复 RL 数据的研究者。","confidence":"high","one_line":["Multi-SWE-bench releases expert-curated multilingual repository issues with Dockerized tests that programmatically score proposed patches.","Multi-SWE-bench 将专家审校的多语言仓库 issue、补丁与 Docker 测试环境打包，以可复现执行评测候选修复。"],"why":"It exposes a multilingual, executable patch-verification surface instead of treating Python-only benchmark scores as general coding ability.","primary_link":"https://arxiv.org/abs/2504.02605","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ByteDance-Seed/Multi-SWE-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://multi-swe-bench.github.io/"}],"link_count":5,"sections":9},{"id":"multichallenge-multi-turn-conversations-2025","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","year":2025,"venue":"Findings of ACL 2025","authors":["Ved Sirdeshmukh","Kaustubh Deshpande","Johannes Mols","Lifeng Jin","Ed-Yeremai Cardona","Dean Lee","Jeremy Kritz","Willow Primack","Summer Yue","Chen Xing"],"authors_zh":"Ved Sirdeshmukh、Kaustubh Deshpande、Johannes Mols、Lifeng Jin、Ed-Yeremai Cardona、Dean Lee、Jeremy Kritz、Willow Primack、Summer Yue、Chen Xing","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["MultiChallenge evaluates instruction retention, inference memory, versioned editing, and self-coherence in multi-turn dialogue with instance-level rubrics.","MultiChallenge 用实例级量表评估多轮对话中的指令保持、推断记忆、版本编辑与自洽性。"],"why":"MultiChallenge evaluates instruction retention, inference memory, versioned editing, and self-coherence in multi-turn dialogue with instance-level rubrics.","primary_link":"https://aclanthology.org/2025.findings-acl.958/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ekwinox117/multi-challenge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScaleAI/MultiChallenge"}],"link_count":3,"sections":9},{"id":"multimodal-rewardbench-2-interleaved-2025","title":"Multimodal RewardBench 2: Evaluating Omni Reward Models for Interleaved Text and Image","year":2025,"venue":"CVPR 2026","authors":["Yushi Hu","Reyhane Askari-Hemmat","Melissa Hall","Emily Dinan","Luke Zettlemoyer","Marjan Ghazvininejad"],"authors_zh":"Yushi Hu, Reyhane Askari-Hemmat, Melissa Hall, Emily Dinan, Luke Zettlemoyer, Marjan Ghazvininejad","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["omni-models","interleaved-generation","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"图文交错 omni 奖励模型的专家偏好评测基准论文","best_for_zh":"需要以统一协议评估图像生成、编辑、交错生成与图像推理奖励模型的研究者。","confidence":"high","one_line":["MMRB2 tests one reward-model interface across image generation, editing, interleaved outputs, and image reasoning using 4,000 expert preference pairs.","四类任务各 1,000 个专家偏好对，覆盖图像生成、编辑、交错生成与图像推理，适合研究统一多模态 judge。"],"why":"It reveals whether a judge can carry one preference contract across both understanding and generation.","primary_link":"https://openaccess.thecvf.com/content/CVPR2026/html/Hu_Multimodal_RewardBench_2_Evaluating_Omni_Reward_Models_for_Interleaved_Text_CVPR_2026_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/MMRB2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/rl-research/multimodal-rewardbench-2"}],"link_count":4,"sections":9},{"id":"multimodal-rewardbench-vlm-2025","title":"Multimodal RewardBench: Holistic Evaluation of Reward Models for Vision Language Models","year":2025,"venue":"arXiv 2025","authors":["Michihiro Yasunaga","Luke Zettlemoyer","Marjan Ghazvininejad"],"authors_zh":"Michihiro Yasunaga, Luke Zettlemoyer, Marjan Ghazvininejad","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["vision-language","reward-modeling","reasoning-evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"视觉语言奖励模型的专家标注评测基准论文","best_for_zh":"需要训练或审计视觉语言奖励模型在正确性、偏好、知识、推理、安全与视觉问答中表现的研究者。","confidence":"high","one_line":["Multimodal RewardBench provides 5,211 expert-annotated multimodal preference triplets for evaluating VLM reward models across six domains.","5,211 个图文提示—优/劣回答三元组，覆盖推理、知识、安全、VQA 等六类反馈面，是训练和审计视觉语言判别器的基础基准。"],"why":"It supplies the image-conditioned chosen/rejected supervision needed to audit VLM judges beyond text-only reward benchmarks.","primary_link":"https://arxiv.org/abs/2502.14191","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/multimodal_rewardbench"}],"link_count":3,"sections":9},{"id":"mutual-taught-2025","title":"Mutual-Taught for Co-adapting Policy and Reward Models","year":2025,"venue":"ACL","authors":["Tianyuan Shi","Canbin Huang","Fanqi Wan","Longguang Zhong","Ziyi Yang","Weizhou Shen","Xiaojun Quan","Ming Yan"],"authors_zh":"Tianyuan Shi、Canbin Huang、Fanqi Wan、Longguang Zhong、Ziyi Yang、Weizhou Shen、Xiaojun Quan、Ming Yan","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["general_reasoning","dialogue"],"tags":["mutual-taught","policy-reward-coadaptation","pseudo-preferences","iterative-dpo","reward-model-shift","self-training","model-selection","low-quality-data-filtering"],"status":"partial","priority":"可读","paper_type_zh":"动态偏好数据构造与策略—奖励模型协同训练配方","best_for_zh":"研究迭代 DPO、奖励模型分布漂移、伪偏好筛选和闭环奖励风险的读者","confidence":"high","one_line":["Alternates RM-ranked on-policy DPO with mixed reward-model self-training, using filtered updated-versus-previous policy pairs to refresh the RM without new human labels.","交替用当前 RM 排序的 on-policy 偏好训练策略，再用策略更新前后的响应构造经低质量过滤的混合偏好数据刷新 RM；机制可核验，但官方代码链接当前没有 Mutual-Taught 专用实现。"],"why":"Treats preference records, checkpoint selection, and reward-model refresh as one evolving construction pipeline while exposing a circular-label risk that static preference datasets hide.","primary_link":"https://aclanthology.org/2025.acl-long.794/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Stycoo/Mutual-Taught"}],"link_count":5,"sections":9},{"id":"mwpo-preference-strength-length-2025","title":"MWPO: Enhancing LLMs Performance through Multi-Weight Preference Strength and Length Optimization","year":2025,"venue":"Findings of ACL 2025","authors":["Shiyue Xu","Fu Zhang","Jingwei Cheng","Linfeng Zhou"],"authors_zh":"Shiyue Xu、Fu Zhang、Jingwei Cheng、Linfeng Zhou（东北大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","sample-weighting","length-control","dpo"],"status":"verified","priority":"必读","paper_type_zh":"偏好数据加权与优化目标研究","best_for_zh":"适合设计离线偏好优化、且需要同时应对弱偏好标签与长度偏差的读者。","confidence":"high","one_line":["MWPO weights each DPO preference pair by its implicit reward margin and response-length margin to improve alignment while counteracting verbosity.","MWPO 依据隐式奖励差距与回答长度差距为每个 DPO 偏好回答对赋权，以提升对齐表现并抑制冗长输出。"],"why":"It makes preference strength and length bias explicit record-level training signals rather than treating all chosen-rejected pairs equally.","primary_link":"https://aclanthology.org/2025.findings-acl.1057/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AIR-hl/MWPO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuggingFaceH4/ultrafeedback_binarized"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/AIR-hl/Mistral-7B-Base-MWPO"}],"link_count":5,"sections":9},{"id":"naturalreasoning-2025","title":"NATURALREASONING: Reasoning in the Wild with 2.8M Challenging Questions","year":2025,"venue":"arXiv preprint (2025)","authors":["Weizhe Yuan","Jane Yu","Song Jiang","Karthik Padthe","Yang Li","Dong Wang","Ilia Kulikov","Kyunghyun Cho","Yuandong Tian","Jason E Weston","Xian Li"],"authors_zh":"Weizhe Yuan、Jane Yu、Song Jiang、Karthik Padthe、Yang Li、Dong Wang、Ilia Kulikov、Kyunghyun Cho、Yuandong Tian、Jason E Weston、Xian Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English STEM, economics, social science, and other natural-domain reasoning"],"tags":["instruction-demonstration-rationale","arxiv-2502.13124","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"推理监督微调、知识蒸馏和过滤式自训练","confidence":"high","one_line":["NATURALREASONING scales domain-diverse question generation, attaches reference answers and named teacher responses, and exposes those records for distillation or self-training.","NATURALREASONING 把自然领域难题、参考答案和具名教师回答整理为 280 万条可蒸馏记录。"],"why":"Open reasoning supervision is concentrated in math and code, so students see too few difficult questions from ordinary scientific and social domains.","primary_link":"https://arxiv.org/abs/2502.13124","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/natural_reasoning"}],"link_count":2,"sections":9},{"id":"naturalthoughts-2025","title":"NaturalThoughts: Selecting and Distilling Reasoning Traces for General Reasoning Tasks","year":2025,"venue":"arXiv preprint (2025)","authors":["Yang Li","Youssef Emad","Karthik Padthe","Jack Lanchantin","Weizhe Yuan","Thao Nguyen","Jason Weston","Shang-Wen Li","Dong Wang","Ilia Kulikov","Xian Li"],"authors_zh":"Yang Li, Youssef Emad, Karthik Padthe, Jack Lanchantin, Weizhe Yuan, Thao Nguyen, Jason Weston, Shang-Wen Li, Dong Wang, Ilia Kulikov, Xian Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing"],"domains":["reasoning-data","general-stem"],"tags":["instruction-demonstration-rationale","reasoning-trace-selection","arxiv-2507.01921","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"通用推理轨迹选择与蒸馏研究","best_for_zh":"适合为通用 STEM SFT 选择 teacher 轨迹，或比较数据规模与策展效果的读者。","confidence":"high","one_line":["NaturalThoughts selects DeepSeek-R1 traces by scale, difficulty, and reasoning-strategy diversity to identify which demonstrations best transfer general STEM reasoning to smaller students.","NaturalThoughts 按规模、难度和推理策略多样性选择 DeepSeek-R1 轨迹，研究哪些示范最能把通用 STEM 推理迁移给小模型。"],"why":"It separates several reasoning-trace selection variables instead of treating all teacher outputs as equally useful.","primary_link":"https://arxiv.org/abs/2507.01921","links":[],"link_count":1,"sections":9},{"id":"nemotron-cascade-2025","title":"Nemotron-Cascade: Scaling Cascaded Reinforcement Learning for General-Purpose Reasoning Models","year":2025,"venue":"arXiv preprint","authors":["Boxin Wang","Chankyu Lee","Nayeon Lee","Sheng-Chieh Lin","Wenliang Dai","Yang Chen","Yangyi Chen","Zhuolin Yang","Zihan Liu","Mohammad Shoeybi","Bryan Catanzaro","Wei Ping"],"authors_zh":"Boxin Wang 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","competitive_programming","software_engineering","instruction_following","tool_calling","science"],"tags":["nvidia","nemotron-cascade","frontier-report","data-disclosure-ledger","cascade-rl","rlhf","rlvr","reward-model","instruction-following","math","code","software-engineering"],"status":"partial","priority":"必读","paper_type_zh":"级联强化学习技术报告与工件发布","best_for_zh":"审计按领域 RLVR 课程、公开工件与训练谱系边界的读者","confidence":"high","one_line":["Nemotron-Cascade documents and partially releases a two-stage SFT plus RLHF -> IF-RL -> Math -> Code -> SWE cascade, with stage-specific feedback ranging from a scalar RM to tests and patch-similarity judging, but complete data-to-run lineage is unavailable.","Nemotron-Cascade 报告了 SFT 和 RLHF 之后按领域顺序开展的 RL，并提供 NVIDIA 的模型/数据工件，但逐领域反馈合约与端到端审计谱系仍不完整。"],"why":"It lets the frontier-report track audit weights, datasets, feedback contracts, intermediate checkpoints, and benchmark results as separate evidence layers instead of treating a collection page as proof of full reproducibility.","primary_link":"https://arxiv.org/abs/2512.13607","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-Cascade-RL-SWE"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/Nemotron-Cascade-8B-Thinking"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/nemotron-cascade"}],"link_count":6,"sections":9},{"id":"nemotron-crossthink-2025","title":"Nemotron-CrossThink: Scaling Self-Learning beyond Math Reasoning","year":2025,"venue":"arXiv preprint","authors":["Syeda Nahida Akter","Shrimai Prabhumoye","Matvei Novikov","Seungju Han","Ying Lin","Evelina Bakhturina","Eric Nyberg","Yejin Choi","Mostofa Patwary","Mohammad Shoeybi","Bryan Catanzaro"],"authors_zh":"Syeda Nahida Akter 等","tracks":["frontier_reports_data_disclosure_ledger","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","science","general_reasoning"],"tags":["nvidia","nemotron-crossthink","frontier-report","data-disclosure-ledger","rlvr","verifiable-rewards","data-blending"],"status":"partial","priority":"必读","paper_type_zh":"跨领域可验证强化学习技术报告","best_for_zh":"希望审计数学以外 RLVR 数据来源、答案格式和反馈披露边界的读者","confidence":"high","one_line":["Nemotron-CrossThink builds a 588,645-prompt multi-domain RLVR mixture with template and answer-space filters, releases 287,376 synthetic QA/math prompt records, and trains Qwen2.5 policies with exact answer-plus-format GRPO rewards.","Nemotron-CrossThink 报告了采用结构化、可验证问答和优化数据混合的跨领域 RL，但未披露语料清单、验证器实现或可复现控制信息。"],"why":"For frontier_reports_data_disclosure_ledger, it is a concrete case where the paper, public synthetic subsets, reward formula, blend study, and training settings are visible, while the exact mixed training manifest, online rollout/reward lineage, and record-level provenance remain incomplete.","primary_link":"https://arxiv.org/abs/2504.13941","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-CrossThink"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/adlr/Nemotron-CrossThink/"}],"link_count":6,"sections":9},{"id":"nemotron-math-v2-2025","title":"Nemotron-Math: Efficient Long-Context Distillation of Mathematical Reasoning from Multi-Mode Supervision","year":2025,"venue":"arXiv preprint (2025)","authors":["Du, Wei","Toshniwal, Shubham","Kisacanin, Branislav","Mahdavi, Sadegh","Moshkov, Ivan","Armstrong, George","Ge, Stephen","Minasyan, Edgar","Chen, Feng","Gitman, Igor"],"authors_zh":"Du, Wei、Toshniwal, Shubham、Kisacanin, Branislav、Mahdavi, Sadegh、Moshkov, Ivan、Armstrong, George、Ge, Stephen、Minasyan, Edgar、Chen, Feng、Gitman, Igor","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English competition and community mathematics, with optional Python tool use"],"tags":["instruction-demonstration-rationale","arxiv-2512.15489","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"长上下文数学推理监督微调","confidence":"high","one_line":["Nemotron-Math generates six supervision modes per source family and packages expected answers and provenance for long-context distillation.","Nemotron-Math 为奥数与社区数学题生成 750 万条多推理预算、可选 Python 工具的长解答。"],"why":"Existing math supervision rarely combines several reasoning budgets, very long traces, community questions, and tool-integrated variants under one comparable schema.","primary_link":"https://arxiv.org/abs/2512.15489","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-Math-v2"}],"link_count":2,"sections":9},{"id":"nemotron-post-training-dataset-v2-2025","title":"Nemotron-Post-Training-Dataset-v2","year":2025,"venue":"arXiv","authors":["Dhruv Nathawani","Shuoyang Ding","Vitaly Lavrukhin","Igor Gitman","Somshubra Majumdar","Evelina Bakhturina","Boris Ginsburg","Jane Polak Scowcroft"],"authors_zh":"Dhruv Nathawani、Shuoyang Ding、Vitaly Lavrukhin、Igor Gitman、Somshubra Majumdar、Evelina Bakhturina、Boris Ginsburg、Jane Polak Scowcroft","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["math","code","science","chat","multilingual","instruction_following"],"tags":["nemotron","post-training","multilingual","synthetic-data","sft","gated-release","license-lineage"],"status":"partial","priority":"必读","paper_type_zh":"多领域、多语言后训练数据集发布与构造披露","best_for_zh":"构建多领域或多语言 SFT 混合数据、审计合成数据来源与许可，或研究发布元数据如何衔接 verifier 和训练阶段的读者","confidence":"high","one_line":["Nemotron-Post-Training-Dataset-v2 releases 6,341,414 SFT-style records across technical reasoning, chat, and five target languages, with generator and license metadata but without row-level selection or RL-stage lineage.","Nemotron-Post-Training-Dataset-v2 发布 6,341,414 条覆盖技术推理、对话和五种目标语言的 SFT 风格记录，并保留生成器与许可字段；但逐条来源、筛选判定、拒绝候选、verifier 输出和训练阶段映射仍未披露。"],"why":"For the Data Construction & Open Release Recipes track, it is both a reusable SFT source and a concrete audit case showing that detailed report-level verifiers and post-training stages do not substitute for record-level provenance, decisions, and run mappings.","primary_link":"https://arxiv.org/abs/2508.14444","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-Post-Training-Dataset-v2"}],"link_count":3,"sections":9},{"id":"agent-sft-2025","title":"Nex-N1: Agentic Models Trained via a Unified Ecosystem for Large-Scale Environment Construction","year":2025,"venue":"arXiv preprint (2025)","authors":["Nex-AGI Team"],"authors_zh":"Nex-AGI Team","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English coding, research, chat, HTML generation, and tool-using agents"],"tags":["instruction-demonstration-rationale","arxiv-2512.04987","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"代理监督微调","confidence":"high","one_line":["Nex Agent-SFT reselects tasks from the Nex ecosystem, regenerates responses with a tool-capable teacher, and publishes six splits under one message-and-tools contract.","Nex Agent-SFT 把 69008 条代码、研究、对话、网页和工具轨迹统一为消息与工具结构。"],"why":"Agent data is fragmented by environment and schema, making it hard to train one model across coding, research, browsing, and tool calling.","primary_link":"https://arxiv.org/abs/2512.04987","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nex-agi/agent-sft"}],"link_count":2,"sections":9},{"id":"no-free-labels-llm-judge-grounding-2025","title":"No Free Labels: Limitations of LLM-as-a-Judge Without Human Grounding","year":2025,"venue":"arXiv","authors":["Michael Krumdick","Charles Lovering","Varshini Reddy","Seth Ebner","Chris Tanner"],"authors_zh":"Michael Krumdick, Charles Lovering, Varshini Reddy, Seth Ebner, Chris Tanner","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["finance","general"],"tags":["llm_as_judge","human_grounding","finance","benchmark","correctness"],"status":"verified","priority":"可读","paper_type_zh":"专家正确性裁决数据与 LLM-as-a-Judge 人类依据评测","best_for_zh":"需要验证自动评审器是否真正判断正确性、而非仅复现风格偏好的研究者。","confidence":"high","one_line":["No Free Labels shows that LLM judges need human grounding to reliably grade correctness beyond questions they can answer.","No Free Labels 以金融专家的参考答案和正确性裁决揭示：没有人类依据时，LLM Judge 难以稳定判断正确性。"],"why":"It turns a common evaluator assumption into a testable limitation with expert reference and judgment data.","primary_link":"https://arxiv.org/abs/2503.05061","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/michaelkrumdickkensho/BFF-Bench"}],"link_count":3,"sections":9},{"id":"nvidia-nemotron-3-2025","title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","year":2025,"venue":"arXiv preprint","authors":["NVIDIA"],"authors_zh":"NVIDIA","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","preference_learning","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","science","coding","software_engineering","search","general","agentic_tool_use","long_context"],"tags":["nvidia","nemotron-3","frontier-report","data-disclosure-ledger","multi-environment-rl","grpo","open-weights","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型家族白皮书与数据披露账本","best_for_zh":"希望区分公开、门控与私有训练材料，并审计多环境 RL 复现边界的研究者","confidence":"high","one_line":["Nemotron 3 documents multi-environment RL and broad weights/data/recipe releases, while retaining proprietary or gated source boundaries and incomplete per-environment audit detail.","Nemotron 3 披露多环境 RL 以及广泛的权重、数据和 recipe 发布，但仍保留专有或门控来源边界与不完整的逐环境审计细节。"],"why":"It separates official artifacts and reference recipes from the missing source allocations, environment pins, rollout logs, reward calibration, and rights mapping needed to reproduce the full family.","primary_link":"https://arxiv.org/abs/2512.20856","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Nemotron"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/nemotron/Nemotron-3/"}],"link_count":5,"sections":9},{"id":"nvidia-nemotron-nano-2-2025","title":"NVIDIA Nemotron Nano 2: An Accurate and Efficient Hybrid Mamba-Transformer Reasoning Model","year":2025,"venue":"arXiv preprint","authors":["NVIDIA"],"authors_zh":"NVIDIA（集体署名）","tracks":["frontier_reports_data_disclosure_ledger","data_construction_open_release_recipes"],"source_role":["model_report","data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","distillation","preference_learning","rlvr","agent_training","safety_alignment"],"construction_layer":["frontier_pipeline","trace_writing"],"domains":["general_reasoning","mathematics","code","science","tool_use","multilingual","safety"],"tags":["nvidia","nemotron-nano-2","reasoning-traces","tool-use","sft","dpo","grpo","rlhf","distillation","open-data"],"status":"verified","priority":"必读","paper_type_zh":"开放推理模型数据、对齐与压缩技术报告","best_for_zh":"研究推理轨迹构造、工具验证、偏好优化、思考预算与蒸馏复现的读者","confidence":"high","one_line":["Nemotron Nano 2 aligns a 12B hybrid model with about 80B SFT tokens, rule- and reward-model feedback, on-policy DPO, GRPO and RLHF, then prunes/distills it to 9B while releasing the models and most training data.","Nemotron Nano 2 使用约 80B token 的 SFT、规则与奖励模型反馈、在线 DPO、GRPO 和 RLHF 对齐 12B 模型，再剪枝蒸馏至 9B，并发布模型与大部分训练数据。"],"why":"It connects concrete data records, verifiers, optimization stages and compression recovery, and exposes enough released artifacts to audit much of the reasoning-data pipeline.","primary_link":"https://arxiv.org/abs/2508.14444","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Nemotron"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-Post-Training-Dataset-v2"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2"}],"link_count":6,"sections":9},{"id":"nemotron-nano-v2-vl-2025","title":"NVIDIA Nemotron Nano V2 VL","year":2025,"venue":"arXiv preprint","authors":["NVIDIA"],"authors_zh":"NVIDIA","tracks":["frontier_reports_data_disclosure_ledger","instruction_demonstration_rationale_data"],"source_role":["model_report","data_release"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","frontier_pipeline","release_audit"],"domains":["vision_language","documents","video","reasoning","code","long_context"],"tags":["frontier-report","nvidia","multimodal","document-understanding","video","long-context","reasoning-traces","synthetic-data","sft","partial-release","disclosure-ledger"],"status":"partial","priority":"可读","paper_type_zh":"多模态模型技术报告与数据披露账本","best_for_zh":"审计部分共享声明、视觉语言训练配方与公开 checkpoint 边界的读者","confidence":"medium","one_line":["NVIDIA Nemotron Nano V2 VL discloses a five-stage multimodal SFT schedule, weights, an 8.15M-row dataset, and OCR tooling, but the public release is not reconciled with the 39.49M-sample inventory or exact stage lineage.","NVIDIA Nemotron Nano V2 VL 报告混合 Mamba-Transformer VLM、token reduction 和部分数据/配方/代码共享，但具体共享范围及反馈审计工件仍为 unknown。"],"why":"It shows how to audit a frontier report by separating reported training consumption, model-card inventory, released artifacts, and evaluation evidence instead of treating a broad release statement or benchmark result as a complete data recipe.","primary_link":"https://arxiv.org/abs/2511.03929","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Curator/tree/experimental/experimental/nvpdftex"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/Nemotron-VLM-Dataset-v2"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16"},{"key":"project","label":["Project","项目主页"],"url":"https://research.nvidia.com/labs/adlr/files/NVIDIA-Nemotron-Nano-V2-VL-report.pdf"}],"link_count":7,"sections":9},{"id":"off-policy-corrected-rm-2025","title":"Off-Policy Corrected Reward Modeling for Reinforcement Learning from Human Feedback","year":2025,"venue":"COLM 2025","authors":["Johannes Ackermann","Takashi Ishida","Masashi Sugiyama"],"authors_zh":"Johannes Ackermann、Takashi Ishida、Masashi Sugiyama","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["reward_modeling","preference_learning"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["summarization","dialogue","preference_alignment"],"tags":["ocrm","off-policy-correction","reward-model-refresh","importance-weighting","rlhf","distribution-shift"],"status":"verified","priority":"必读","paper_type_zh":"离策略奖励模型校正与刷新配方","best_for_zh":"关注RLHF分布漂移、重要性加权、奖励模型刷新和离线偏好数据的研究者","confidence":"high","one_line":["Retrains reward models on one fixed preference dataset using stage-specific current-to-SFT importance weights between blocks of policy optimization.","在 PPO 阶段之间，用当前策略相对 SFT 行为策略的重要性权重重训奖励模型；278,496 条偏好记录与标签不变，改变的是有效训练分布。"],"why":"Separates preference-data acquisition from effective-distribution construction and makes the support, variance, behavior-likelihood, cadence, and evaluator assumptions of off-policy reward-model correction auditable.","primary_link":"https://openreview.net/forum?id=0zxugBcgF5","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JohannesAck/OffPolicyCorrectedRewardModeling"}],"link_count":4,"sections":9},{"id":"olmo-2-32b-2025","title":"OLMo 2 32B: First fully open model to outperform GPT 3.5 and GPT 4o mini","year":2025,"venue":"Ai2 technical release","authors":["Ai2"],"authors_zh":"Ai2","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","data_release","construction_recipe"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general","mathematics","instruction_following"],"tags":["ai2","allenai","olmo-2","olmo-2-32b","frontier-report","data-disclosure-ledger","sft","dpo","rlvr","grpo","open-data"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型发布说明与数据披露账本","best_for_zh":"需要审计开放权重模型的阶段化数据、RLVR 反馈边界与来源权利的研究者","confidence":"high","one_line":["Ai2's OLMo 2 32B release links its data, weights, and recipe, and discloses SFT filtering, five-completion math majority selection, on-policy preference data, and GRPO-RLVR task families, while leaving reward implementation and complete provenance/audit details unresolved.","Ai2 的 OLMo 2 32B 发布链接了数据、权重和 recipe，并披露 SFT 过滤、五次 completion 的数学多数选择、on-policy 偏好数据和 GRPO-RLVR 任务族；reward 实现与完整来源/审计细节仍未解决。"],"why":"It is an unusually inspectable frontier model release: readers can distinguish released mixtures and checkpoints from the still-unresolved generator, reward, provenance, licence, and contamination details required to reproduce or audit the entire 32B post-training path.","primary_link":"https://allenai.org/blog/olmo2-32b","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/OLMo-core"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/tulu-3-sft-olmo-2-mixture-0225"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/allenai/OLMo-2-0325-32B-Instruct"},{"key":"project","label":["Project","项目主页"],"url":"https://allenai.org/olmo"}],"link_count":5,"sections":9},{"id":"alphaproof-olympiad-level-formal-mathematical-reasoning-2025","title":"Olympiad-level formal mathematical reasoning with reinforcement learning","year":2025,"venue":"Nature 651 (2026)","authors":["Thomas Hubert","Rishi Mehta","Laurent Sartran","Miklós Z. Horváth","Goran Žužić","Eric Wieser","Aja Huang","Julian Schrittwieser","Yannick Schroecker","Hussain Masoom","Ottavia Bertolli","Tom Zahavy","Amol Mandhane","Jessica Yung","Iuliya Beloshapka","Borja Ibarz","Vivek Veeriah","Lei Yu","Oliver Nash","Paul Lezeau","Salvatore Mercuri","Calle Sönne","Bhavik Mehta","Alex Davies","Daniel Zheng","Fabian Pedregosa","Yin Li","Ingrid von Glehn","Mark Rowland","Samuel Albanie","Ameya Velingker","Simon Schmitt","Edward Lockhart","Edward Hughes","Henryk Michalewski","Nicolas Sonnerat","Demis Hassabis","Pushmeet Kohli","David Silver"],"authors_zh":"Thomas Hubert, Rishi Mehta, Laurent Sartran, Miklós Z. Horváth, Goran Žužić, Eric Wieser, Aja Huang, Julian Schrittwieser, Yannick Schroecker, Hussain Masoom, Ottavia Bertolli, Tom Zahavy, Amol Mandhane, Jessica Yung, Iuliya Beloshapka, Borja Ibarz, Vivek Veeriah, Lei Yu, Oliver Nash, Paul Lezeau, Salvatore Mercuri, Calle Sönne, Bhavik Mehta, Alex Davies, Daniel Zheng, Fabian Pedregosa, Yin Li, Ingrid von Glehn, Mark Rowland, Samuel Albanie, Ameya Velingker, Simon Schmitt, Edward Lockhart, Edward Hughes, Henryk Michalewski, Nicolas Sonnerat, Demis Hassabis, Pushmeet Kohli, David Silver","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","programmatic_code_formal"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["Olympiad-level formal mathematical reasoning with reinforcement learning records Lean tactic-state search and verified proof/disproof outcomes under Lean checker feedback.","AlphaProof 在 Lean 中以树搜索和测试时强化学习构造可验证证明经验，但核心课程与搜索日志未公开。"],"why":"It makes budgeted tree search with replay of discovered proofs/disproofs and its audit boundary visible for reasoning-data curation.","primary_link":"https://www.nature.com/articles/s41586-025-09833-y","links":[],"link_count":2,"sections":9},{"id":"omnialign-v-human-preference-2025","title":"OmniAlign-V: Towards Enhanced Alignment of MLLMs with Human Preference","year":2025,"venue":"ACL 2025","authors":["Xiangyu Zhao","Shengyuan Ding","Zicheng Zhang","Haian Huang","Maosong Cao","Weiyun Wang","Jiaqi Wang","Xinyu Fang","Wenhai Wang","Guangtao Zhai","Haodong Duan","Hua Yang","Kai Chen"],"authors_zh":"Xiangyu Zhao、Shengyuan Ding、Zicheng Zhang 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-alignment","vision-language","preference-learning"],"tags":["multimodal","preference-pairs","dpo","human-alignment","vision-language"],"status":"verified","priority":"可读","paper_type_zh":"多模态偏好数据集、构建流程与对齐评测论文","best_for_zh":"需要复用图像条件偏好对、审计 AI 评审负样本，或比较多模态 SFT 与 DPO 的研究者。","confidence":"high","one_line":["OmniAlign-V-DPO releases 150K image-grounded chosen-rejected pairs built from OmniAlign-V answers and model-sampled negative responses for MLLM preference optimization.","OmniAlign-V-DPO 将图像、开放式问题、优选回答和模型采样的劣选回答发布为 15 万条偏好对，用于多模态大语言模型的 DPO 对齐。"],"why":"It makes the preference contract explicit for multimodal alignment: a detailed image-grounded answer is positive, while an intent-divergent sampled response is negative.","primary_link":"https://arxiv.org/abs/2502.18411","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PhoenixZ810/OmniAlign-V"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PhoenixZ/OmniAlign-V-DPO"},{"key":"project","label":["Project","项目主页"],"url":"https://phoenixz810.github.io/OmniAlign-V/"}],"link_count":5,"sections":9},{"id":"evaluating-alignment-through-judges-2025","title":"On Evaluating LLM Alignment by Evaluating LLMs as Judges","year":2025,"venue":"NeurIPS 2025 (Poster)","authors":["Yixin Liu","Pengfei Liu","Arman Cohan"],"authors_zh":"Yixin Liu, Pengfei Liu, Arman Cohan","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"对齐评测基准与生成—评测一致性研究","best_for_zh":"需要低成本、可复用人类偏好对齐评测的研究者。","confidence":"high","one_line":["Connects generation-evaluation consistency with alignment assessment of LLM judges.","以 LLM 作为判别器的能力间接评测生成对齐。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=OBaK9JSbHk","links":[],"link_count":1,"sections":9},{"id":"llm-judge-code-summarization-2025","title":"On the Effectiveness of LLM-as-a-judge for Code Generation and Summarization","year":2025,"venue":"IEEE Transactions on Software Engineering","authors":["Giuseppe Crupi","Rosalia Tufano","Alejandro Velasco","Antonio Mastropaolo","Denys Poshyvanyk","Gabriele Bavota"],"authors_zh":"Giuseppe Crupi, Rosalia Tufano, Alejandro Velasco, Antonio Mastropaolo, Denys Poshyvanyk, Gabriele Bavota","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["llm-as-a-judge","code-evaluation","audit"],"status":"verified","priority":"可读","paper_type_zh":"大语言模型评审可靠性实证研究","best_for_zh":"需要校准代码自动评测或代码奖励信号的研究者。","confidence":"medium","one_line":["Tests whether LLM judges reliably assess code correctness and code-summary quality against executable and human references.","审计大语言模型评审代码正确性和代码摘要质量时的可靠性与失效模式。"],"why":"It quantifies judge failures that can make automated code evaluation or reward signals unsafe.","primary_link":"https://arxiv.org/abs/2507.16587","links":[],"link_count":2,"sections":9},{"id":"long-cot-collection-2025","title":"One Missing Piece for Open-Source Reasoning Models: A Dataset to Mitigate Cold-Starting Short CoT LLMs in RL","year":2025,"venue":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 6: Industry Track)","authors":["Hyungjoo Chae","Dongjin Kang","Jihyuk Kim","Beong-woo Kwak","Sunghyun Park","Haeju Park","Jinyoung Yeo","Moontae Lee","Kyungjae Lee"],"authors_zh":"Hyungjoo Chae、Dongjin Kang、Jihyuk Kim、Beong-woo Kwak、Sunghyun Park、Haeju Park、Jinyoung Yeo、Moontae Lee、Kyungjae Lee","tracks":["rollout_search_test_time_trace_data","data_construction_open_release_recipes"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical_reasoning","general_reasoning"],"tags":["long-cot-collection","long-cot","short-cot","thought-budget","cold-start","sft","rlvr","gpt-4o","o1","trace-construction"],"status":"partial","priority":"可读","paper_type_zh":"长链推理轨迹构造与 RLVR 冷启动配方","best_for_zh":"研究长推理轨迹合成、思考预算控制、答案级过滤、SFT 冷启动与 RLVR 数据审计的读者","confidence":"high","one_line":["A paper-described 100K long-CoT recipe retrieves o1 flow/budget seeds to guide GPT-4o stepwise generation and answer judgment, but the publication's release target remains the literal text LINK and no paper-matched artifact is verified.","Long CoT Collection 以 o1 推理流与思考预算种子检索来引导 GPT-4o 分步生成并进行答案判断；论文声称构造 100K 条数据，但发布位置仍是字面量“LINK”，尚无可核验的论文配套制品。"],"why":"It makes reasoning-flow shape and thought budget explicit construction variables for long-trace SFT before RLVR, while exposing how missing releases and row-level decisions prevent independent audit or reuse.","primary_link":"https://aclanthology.org/2025.acl-industry.85/","links":[],"link_count":3,"sections":9},{"id":"one-token-to-fool-2025","title":"One Token to Fool LLM-as-a-Judge","year":2025,"venue":"arXiv","authors":["Yulai Zhao","Haolin Liu","Dian Yu","S. Y. Kung","Meijia Chen","Haitao Mi","Dong Yu"],"authors_zh":"Yulai Zhao、Haolin Liu、Dian Yu、S. Y. Kung、Haitao Mi、Dong Yu","tracks":["audit_failure_contamination_verifier_attacks","preference_reward_feedback_data"],"source_role":["audit_failure","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","reward_modeling"],"construction_layer":["reward_verifier_layer"],"domains":["judge"],"tags":[],"status":"verified","priority":"必读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Superficial symbols or reasoning openers can make LLM judges award reference-conditioned positive rewards; truncated-lead-in negatives harden Master-RMs.","证明极小的表面线索也能翻转 LLM 裁判结果，提醒量规和评审提示本身需要做对抗性测试。"],"why":"It shows that supplying a reference answer does not make a generative verifier safe to use as an RL reward without adversarial auditing.","primary_link":"https://arxiv.org/abs/2507.08794","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sarosavo/Master-RM"}],"link_count":2,"sections":9},{"id":"open-reasoner-zero-2025","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","year":2025,"venue":"NeurIPS 2025 (Main Conference Track)","authors":["Jingcheng Hu","Yinmin Zhang","Qi Han","Daxin Jiang","Xiangyu Zhang","Heung-Yeung Shum"],"authors_zh":"Jingcheng Hu、Yinmin Zhang、Qi Han、Daxin Jiang、Xiangyu Zhang、Heung-Yeung Shum","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","data_release","model_report","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward","trajectory_value"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematical_reasoning","general_reasoning","logical_reasoning","coding","instruction_following"],"tags":["open-reasoner-zero","reasoner-zero","rlvr","ppo","learned-critic","mathematical-verifier","hard-prompt-mining","open-data","open-models","release-audit","zero-setting-boundary"],"status":"partial","priority":"必读","paper_type_zh":"大规模基础模型 RLVR 配方与开放发布审计","best_for_zh":"研究 Reasoner-Zero、PPO 与 learned critic、终局数学 verifier、hard-prompt mining，以及开放 RL 配方中数据血缘、重复、污染和运行工件缺口","confidence":"high","one_line":["Open-Reasoner-Zero starts PPO directly from Qwen2.5 Base, samples 64 responses per prompt, scores boxed mathematical answers programmatically, and trains a token-value critic, while releasing prompts, code, and weights but withholding exact run trajectories, failures, and immutable manifests.","Open-Reasoner-Zero 不经过 SFT 或蒸馏、直接从 Qwen2.5 Base 启动 PPO，按每题 64 条在线响应分配终局数学奖励并训练 token-value critic；但公开的是 prompt/参考答案、代码和权重，而不是论文运行的 rollout、失败轨迹、奖励或日志。"],"why":"It makes large-scale critic-based Reasoner-Zero training unusually concrete and accessible, yet shows why an open recipe is not a fully auditable data release: source rights, duplication, benchmark overlap, annealing/mixed-domain configs, compute, and exact rollout lineage remain unresolved.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/ed873d79e7c268c020c4b4db13a2812a-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Open-Reasoner-Zero/Open-Reasoner-Zero"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Open-Reasoner-Zero/Open-Reasoner-Zero/tree/3fdd9a07b4fb01e06005e6e74fc56690cde8a341/data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Open-Reasoner-Zero"},{"key":"project","label":["Project","项目主页"],"url":"https://yasminezhang.notion.site/Open-Reasoner-Zero-19e12cf72d418007b9cdebf44b0e7903"}],"link_count":13,"sections":9},{"id":"openai-gpt-4-5-system-card-2025","title":"OpenAI GPT-4.5 System Card","year":2025,"venue":"OpenAI system card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["sft","preference_learning","safety_alignment","evaluation"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["general_reasoning","safety"],"tags":["openai","gpt-4-5","frontier-report","data-disclosure-ledger","system-card","sft","rlhf","safety-evaluation"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露账本","best_for_zh":"审计前沿模型报告中的阶段级披露，以及阻碍独立复核训练数据、反馈、奖励和谱系的缺失工件","confidence":"high","one_line":["OpenAI's GPT-4.5 system card discloses broad public/partnership/in-house data categories, smaller-model-derived alignment data, SFT, RLHF, filtering, and selected safety evaluation protocols, but not reproducible data or feedback contracts.","OpenAI 的 GPT-4.5 系统卡披露了宽泛的公开/合作/内部数据类别、来自小模型的对齐数据、SFT、RLHF、过滤及部分安全评测协议，但没有公开可复现的数据或反馈契约。"],"why":"It is a central Track 12 comparison point: the report offers meaningful stage-level and safety-evaluation disclosure, yet makes visible which missing artifacts prevent independent audit of source provenance, post-training feedback, reward mechanisms, filtering, and lineage.","primary_link":"https://cdn.openai.com/gpt-4-5-system-card-2272025.pdf","links":[],"link_count":2,"sections":9},{"id":"openai-o3-o4-mini-system-card-2025","title":"OpenAI o3 and o4-mini System Card","year":2025,"venue":"OpenAI System Card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["agent_training","safety_alignment"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["reasoning","tool_use","safety"],"tags":["openai","o3","o4-mini","official-system-card","closed-model","frontier-report","data-disclosure-ledger","deliberative-alignment"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型报告与数据披露账本","best_for_zh":"审计闭源推理模型报告中已披露与未披露的后训练数据、反馈和复现边界","confidence":"high","one_line":["OpenAI's o3/o4-mini System Card discloses high-level data-source and filtering claims plus CoT reinforcement learning and safety alignment, while withholding reusable data, feedback contracts, and recipe details.","OpenAI o3/o4-mini System Card 仅披露高层数据来源、过滤、CoT 强化学习与安全对齐主张；训练数据、反馈契约和具体配方仍不可复用。"],"why":"It is a Track 12 disclosure-boundary case: official claims permit an audit of what a frontier report exposes, but not reconstruction of the post-training data or reward pipeline.","primary_link":"https://cdn.openai.com/pdf/2221c875-02dc-4789-800b-e7758f3722c1/o3-and-o4-mini-system-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/o3-o4-mini-system-card/"}],"link_count":2,"sections":9},{"id":"openai-o3-mini-system-card-2025","title":"OpenAI o3-mini System Card","year":2025,"venue":"OpenAI System Card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["safety_alignment","evaluation"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["reasoning","coding","safety"],"tags":["openai","o3-mini","system-card","frontier-report","data-disclosure-ledger","deliberative-alignment","reinforcement-learning","safety-evaluation"],"status":"partial","priority":"必读","paper_type_zh":"闭源前沿模型系统卡与数据披露账本","best_for_zh":"审计闭源推理模型报告中已披露与未披露的后训练数据、反馈和复现边界","confidence":"high","one_line":["OpenAI's o3-mini System Card reports broad pretraining-source categories, data filtering, large-scale reasoning RL, deliberative alignment, and safety evaluations, but not reusable records, the RL reward contract, or a reproducible post-training recipe.","OpenAI o3-mini System Card 披露了宽泛的预训练数据类别、过滤、大规模推理强化学习、deliberative alignment 与安全评测，但没有公开可复用记录、RL 奖励契约或可复现的后训练配方。"],"why":"It is a high-impact Track 12 disclosure boundary: the official report makes some frontier post-training and safety claims auditable while clearly not supplying the data, feedback, lineage, or environment detail needed to reconstruct them.","primary_link":"https://cdn.openai.com/o3-mini-system-card-feb10.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/o3-mini-system-card/"}],"link_count":2,"sections":9},{"id":"opencodeinstruct-2025","title":"OpenCodeInstruct: A Large-scale Instruction Tuning Dataset for Code LLMs","year":2025,"venue":"arXiv preprint (2025)","authors":["Wasi Uddin Ahmad","Aleksander Ficek","Mehrzad Samadi","Jocelyn Huang","Vahid Noroozi","Somshubra Majumdar","Boris Ginsburg"],"authors_zh":"Wasi Uddin Ahmad、Aleksander Ficek、Mehrzad Samadi、Jocelyn Huang、Vahid Noroozi、Somshubra Majumdar、Boris Ginsburg","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English code generation, debugging, algorithms, and programming questions"],"tags":["instruction-demonstration-rationale","arxiv-2504.04030","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"Llama 与 Qwen 系列的代码监督微调","confidence":"high","one_line":["OpenCodeInstruct scales multiple generation algorithms to five million pairs and stores unit tests, execution status, and model judgments beside each solution.","OpenCodeInstruct 为 500 万条代码指令配上测试、执行状态和模型评审，使解答质量可以逐条检查。"],"why":"Public code SFT sets are much smaller than pretraining corpora and often omit tests and judgments needed to distinguish executable solutions from plausible text.","primary_link":"https://arxiv.org/abs/2504.04030","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenCodeInstruct"}],"link_count":2,"sections":9},{"id":"opencodereasoning-advancing-data-distillation-for-competitive-coding-2025","title":"OpenCodeReasoning: Advancing Data Distillation for Competitive Coding","year":2025,"venue":"COLM 2025","authors":["Wasi Uddin Ahmad","Sean Narenthiran","Somshubra Majumdar","Aleksander Ficek","Siddhartha Jain","Jocelyn Huang","Vahid Noroozi","Boris Ginsburg"],"authors_zh":"Wasi Uddin Ahmad、Sean Narenthiran、Somshubra Majumdar、Aleksander Ficek、Siddhartha Jain、Jocelyn Huang、Vahid Noroozi、Boris Ginsburg","tracks":["data_construction_open_release_recipes","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit","scaling_report"],"domains":["code","competitive_programming","python"],"tags":["opencodereasoning","competitive-programming","deepseek-r1-distillation","code-sft","execution-filtering","data-scaling"],"status":"verified","priority":"必读","paper_type_zh":"竞赛编程推理数据发布、蒸馏配方与执行过滤消融","best_for_zh":"构建代码推理 SFT 数据、研究教师蒸馏与执行筛选，或审计开放数据版本、许可和污染控制的读者","confidence":"high","one_line":["OpenCodeReasoning releases DeepSeek-R1 competitive-programming traces for Qwen2.5 SFT and shows a diversity-versus-execution-filtering trade-off, but the main rows are not functionally certified and the current recipe has moved toward OCR-2.","OpenCodeReasoning 将 DeepSeek-R1 的竞赛编程推理与 Python 解答蒸馏为 Qwen2.5 的 SFT 数据，并揭示多样性与执行过滤的取舍；但主发布集只通过格式和语法检查，当前官方配方也已转向 OCR-2。"],"why":"It makes a large code-reasoning distillation pipeline inspectable and provides a concrete negative result about execution-only selection, while exposing why row-level verification, rights, decontamination logs, and revision pinning remain necessary.","primary_link":"https://arxiv.org/abs/2504.01943","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Skills/tree/main/recipes/opencodereasoning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenCodeReasoning"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/nvidia/opencodereasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://nvidia-nemo.github.io/Skills/releases/opencodereasoning/"}],"link_count":8,"sections":9},{"id":"opencua-2025","title":"OpenCUA: Open Foundations for Computer-Use Agents","year":2025,"venue":"NeurIPS 2025 Main Conference Track","authors":["Xinyuan Wang","Bowen Wang","Dunjie Lu","Junlin Yang","Tianbao Xie","Junli Wang","Jiaqi Deng","Xiaole Guo","Yiheng Xu","Chen Henry Wu","Zhennan Shen","Zhuokai Li","Ryan Li","Xiaochuan Li","Junda Chen","Boyuan Zheng","Peihang Li","Fangyu Lei","Ruisheng Cao","Yeqiao Fu","Dongchan Shin","Martin Shin","Jiarui Hu","Yuyan Wang","Jixuan Chen","Yuxiao Ye","Danyang Zhang","Dikang Du","Hao Hu","Huarong Chen","Zaida Zhou","Haotian Yao","Ziwei Chen","Qizheng Gu","Yipu Wang","Heng Wang","Diyi Yang","Victor Zhong","Flood Sung","Y. Charles","Zhilin Yang","Tao Yu"],"authors_zh":"Xinyuan Wang 等 42 位作者","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","frontier_pipeline"],"domains":["computer_use","multimodal_agents"],"tags":["opencua","agentnet","computer-use-agents","multimodal-trajectories","human-demonstrations","reflective-cot","state-action-alignment","agentnetbench","supervised-fine-tuning","release-version-drift"],"status":"partial","priority":"必读","paper_type_zh":"跨平台计算机使用轨迹数据与开放构造配方","best_for_zh":"研究 computer-use agent、跨平台人类演示、多模态轨迹构造、反思式监督与数据审计的读者","confidence":"high","one_line":["OpenCUA turns 22,625 human desktop tasks into aligned screenshot-action traces with Claude-generated reflective supervision and releases the data, models, tools, and offline evaluator, while leaving full training and environment replay incomplete.","OpenCUA 将 22,625 个人类桌面任务处理为对齐的截图—动作轨迹并加入 Claude 合成的反思式监督，已发布数据、模型、工具和离线评测器，但完整训练与环境复现仍不完备。"],"why":"For the Data Construction track, it makes collection, action reduction, visual-state matching, synthetic reflection, filtering, source mixing, and deployment evaluation separable audit objects instead of treating model benchmark gains as data quality.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/cc7ae529e945226b0d52ea4ac478c4f3-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xlang-ai/OpenCUA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xlangai/AgentNet"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/xlangai/opencua-open-foundations-for-computer-use-agents-6882014ebecdbbe46074a68d"},{"key":"project","label":["Project","项目主页"],"url":"https://opencua.xlang.ai/"}],"link_count":8,"sections":9},{"id":"openmathinstruct-2-2025","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","year":2025,"venue":"ICLR 2025","authors":["Shubham Toshniwal","Wei Du","Ivan Moshkov","Branislav Kisacanin","Alexan Ayrapetyan","Igor Gitman"],"authors_zh":"Shubham Toshniwal、Wei Du、Ivan Moshkov、Branislav Kisacanin、Alexan Ayrapetyan、Igor Gitman","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematics"],"tags":["openmathinstruct-2","math-sft","synthetic-instruction-data","llama-3-1-405b","solution-augmentation","problem-augmentation","majority-vote","decontamination","data-scaling"],"status":"partial","priority":"必读","paper_type_zh":"大规模开放数学 SFT 数据发布、构造 recipe 与 scaling study","best_for_zh":"构建数学 reasoning SFT 数据、比较 teacher/format/diversity/scale、实现答案聚合与去污染，或审计开放数据版本和逐条 lineage 的读者","confidence":"high","one_line":["OpenMathInstruct-2 packages original and augmented GSM8K/MATH problems with teacher-written solutions and source-or-majority proxy answers, alongside reproducible generation, decontamination, downsampling, and SFT recipes.","OpenMathInstruct-2 用 Llama-3.1-405B-Instruct 的 solution augmentation 与 question-solution augmentation 构造 13,972,791 条数学 SFT 记录，并用 32 个解答的表面多数票生成新题答案；其规模与开放 recipe 价值明确，但阈值为 0、逐条投票/lineage 缺失和残余 Omni-MATH overlap 限制了可审计性。"],"why":"It is a high-scale worked example of how open reasoning demonstrations are sourced, expanded, answer-filtered, and trained, and it shows why majority answers and dataset-level licenses still need vote, rejection, contamination, and provenance ledgers for trustworthy reuse.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/file/302ce0673c00aee2cf84bb43d0117553-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Skills"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenMathInstruct-2"},{"key":"project","label":["Project","项目主页"],"url":"https://nvidia-nemo.github.io/Skills/openmathinstruct2/"}],"link_count":8,"sections":9},{"id":"openthoughts-data-recipes-2025","title":"OpenThoughts: Data Recipes for Reasoning Models","year":2025,"venue":"ICLR 2026 Oral","authors":["Etash Guha","Ryan Marten","Sedrick Keh","Negin Raoof","Georgios Smyrnis","Hritik Bansal","Marianna Nezhurina","Jean Mercat","Trung Vu","Zayne Sprague","Ashima Suvarna","Benjamin Feuer","Leon Liangyu Chen","Zaid Khan","Eric Frankel","Sachin Grover","Caroline Choi","Niklas Muennighoff","Shiye Su","Wanjia Zhao","John Yang","Shreyas Pimpalgaonkar","Kartik Sharma","Charlie Cheng-Jie Ji","Yichuan Deng","Sarah Pratt","Vivek Ramanujan","Jon Saad-Falcon","Stutee Acharya","Jeffrey Li","Achal Dave","Alon Albalak","Kushal Arora","Blake Wulfe","Chinmay Hegde","Greg Durrett","Sewoong Oh","Mohit Bansal","Saadia Gabriel","Aditya Grover","Kai-Wei Chang","Vaishaal Shankar","Aaron Gokaslan","Mike A. Merrill","Tatsunori Hashimoto","Yejin Choi","Jenia Jitsev","Reinhard Heckel","Maheswaran Sathiamoorthy","Alexandros G. Dimakis","Ludwig Schmidt"],"authors_zh":"Etash Guha、Ryan Marten、Sedrick Keh、Negin Raoof、Georgios Smyrnis、Hritik Bansal、Marianna Nezhurina、Jean Mercat、Trung Vu、Zayne Sprague、Ashima Suvarna、Benjamin Feuer、Leon Liangyu Chen、Zaid Khan、Eric Frankel、Sachin Grover、Caroline Choi、Niklas Muennighoff、Shiye Su、Wanjia Zhao、John Yang、Shreyas Pimpalgaonkar、Kartik Sharma、Charlie Cheng-Jie Ji、Yichuan Deng、Sarah Pratt、Vivek Ramanujan、Jon Saad-Falcon、Stutee Acharya、Jeffrey Li、Achal Dave、Alon Albalak、Kushal Arora、Blake Wulfe、Chinmay Hegde、Greg Durrett、Sewoong Oh、Mohit Bansal、Saadia Gabriel、Aditya Grover、Kai-Wei Chang、Vaishaal Shankar、Aaron Gokaslan、Mike A. Merrill、Tatsunori Hashimoto、Yejin Choi、Jenia Jitsev、Reinhard Heckel、Maheswaran Sathiamoorthy、Alexandros G. Dimakis、Ludwig Schmidt","tracks":["data_construction_open_release_recipes","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","scaling_report","release_audit"],"domains":["math","code","science"],"tags":["openthoughts","reasoning-data","open-data","teacher-distillation","question-filtering","repeated-sampling","sft","data-scaling","decontamination","license-risk","secret-hygiene"],"status":"partial","priority":"必读","paper_type_zh":"开放推理数据发布、构造 recipe 与 SFT scaling study","best_for_zh":"设计 reasoning-data 构造管线、比较 prompt/answer 过滤策略、复现 teacher distillation，或审计开放数据许可、provenance、去污染与 secret hygiene 的读者","confidence":"high","one_line":["OpenThoughts releases a 1.2M-row math/code/science SFT corpus and a controlled recipe showing that source choice, prompt filtering, repeated teacher sampling, and teacher selection matter more than the tested answer filters, while leaving row-level correctness and rights unresolved.","OpenThoughts 用来源筛选、prompt 过滤、16 次 QwQ-32B 采样和完整 SFT 构成 120 万条数学/代码/科学推理数据；其主要价值是把构造选择变成可对照的 recipe，但答案未做正确性验证，且来源权利、逐条 provenance 与仓库密钥卫生问题阻断直接训练复用。"],"why":"It turns reasoning-data curation into an experimentally inspectable pipeline with unusually useful negative results, but also demonstrates that open code, data, and model weights do not by themselves establish safe reuse without source-level rights, provenance, rejection logs, and secret-safe configs.","primary_link":"https://openreview.net/forum?id=mbqvBA12Dx","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/open-thoughts/open-thoughts"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/open-thoughts/OpenThoughts3-1.2M"},{"key":"project","label":["Project","项目主页"],"url":"https://www.open-thoughts.ai/"}],"link_count":10,"sections":9},{"id":"openturingbench-2025","title":"OpenTuringBench: An Open-Model-based Benchmark and Framework for Machine-Generated Text Detection and Attribution","year":2025,"venue":"EMNLP 2025","authors":["Lucio La Cava","Andrea Tagarelli"],"authors_zh":"Lucio La Cava, Andrea Tagarelli","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"机器生成文本检测与作者归因基准论文","best_for_zh":"需要评估机器文本检测、归因或训练污染风险的研究者。","confidence":"medium","one_line":["Open benchmark relevant to attribution and audit of generated text under model shift.","以七类压力场景检验开源 LLM 生成文本的检测与作者归因鲁棒性。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://aclanthology.org/2025.emnlp-main.1354/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MLNTeam-Unical/OpenTuringBench"}],"link_count":3,"sections":9},{"id":"openai-operator-system-card-2025","title":"Operator System Card","year":2025,"venue":"OpenAI System Card","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["agent_training"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["agentic_tool_use","computer_use","safety"],"tags":["openai","operator","cua","computer-use","agentic-tool-use","system-card","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型系统卡与数据披露账本","best_for_zh":"需要审计闭源计算机使用智能体报告中已披露与未披露的后训练数据、反馈和复现边界的读者","confidence":"high","one_line":["OpenAI's Operator System Card reports specialized supervised computer-use data, broad public-data sources, and reinforcement learning for a GPT-4o-based CUA, but leaves task records, rewards/verifiers, training environments, and lineage undisclosed.","OpenAI 的《Operator System Card》披露：基于 GPT-4o 的 CUA 使用专门的计算机操作监督数据、广义公开数据来源和强化学习；任务记录、奖励/验证器、训练环境及数据谱系均未披露。"],"why":"It is a clear Track 12 boundary case: the report makes the existence and broad role of computer-use demonstrations and reinforcement learning auditable, yet does not provide enough data, feedback, or environment detail to reconstruct or safely reuse the recipe.","primary_link":"https://cdn.openai.com/operator_system_card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/computer-using-agent/"}],"link_count":3,"sections":9},{"id":"gvm-raft-2025","title":"Optimizing Chain-of-Thought Reasoners via Gradient Variance Minimization in Rejection Sampling and RL","year":2025,"venue":"NeurIPS 2025","authors":["Jiarui Yao","Yifan Hao","Hanning Zhang","Hanze Dong","Wei Xiong","Nan Jiang","Tong Zhang"],"authors_zh":"Jiarui Yao、Yifan Hao、Hanning Zhang、Hanze Dong、Wei Xiong、Nan Jiang、Tong Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematics"],"tags":["gvm-raft","rejection-sampling","dynamic-sample-allocation","gradient-variance","adaptive-rollouts","grpo"],"status":"verified","priority":"可读","paper_type_zh":"数学推理动态 rollout 分配与 RAFT/GRPO 训练配方","best_for_zh":"研究在线拒绝采样、RLVR、动态推理预算、梯度方差与训练数据 lineage 的读者","confidence":"high","one_line":["GVM-RAFT estimates per-prompt Math-Verify acceptance and accepted-response gradient magnitude from pilot rollouts, then reallocates a fixed online sampling budget before RAFT++ or GRPO updates.","GVM-RAFT 用 pilot rollout 估计逐提示 Math-Verify 接受率与被接受回答的梯度幅度，在固定总预算下重分配 RAFT++ 或 GRPO 采样，但论文运行缺少可审计的逐项 lineage 发布。"],"why":"It turns rollout allocation into an auditable construction layer rather than a uniform default, while also showing why reproducing the result requires prompt, verifier, gradient, allocation, accepted/rejected trace, and checkpoint lineage that the code release does not package.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/eef6cb60fd59b32d35718e176b4b08d6-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RLHFlow/GVM"}],"link_count":4,"sections":9},{"id":"optimizing-test-time-compute-meta-reinforcement-finetuning-2025","title":"Optimizing Test-Time Compute via Meta Reinforcement Finetuning","year":2025,"venue":"arXiv","authors":["Yuxiao Qu","Matthew Y. R. Yang","Amrith Setlur","Lewis Tunstall","Edward Emanuel Beeching","Ruslan Salakhutdinov","Aviral Kumar"],"authors_zh":"Yuxiao Qu, Matthew Y. R. Yang, Amrith Setlur, Lewis Tunstall, Edward Emanuel Beeching, Ruslan Salakhutdinov, Aviral Kumar","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning"],"tags":["track5","online_ttc_trace_reuse"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["Optimizing Test-Time Compute via Meta Reinforcement Finetuning records test-time episodes and reasoning fragments under success-probability-change bonus plus outcome reward.","MRT 把思考过程切分为 episode，并以元证明器估计的进展奖励辅助终局强化学习。"],"why":"It makes meta-RL allocates test-time compute and its audit boundary visible for reasoning-data curation.","primary_link":"https://arxiv.org/abs/2503.07572","links":[],"link_count":2,"sections":9},{"id":"os-genesis-2025","title":"OS-Genesis: Automating GUI Agent Trajectory Construction via Reverse Task Synthesis","year":2025,"venue":"ACL 2025","authors":["Qiushi Sun","Kanzhi Cheng","Zichen Ding","Chuanyang Jin","Yian Wang","Fangzhi Xu","Zhenyu Wu","Chengyou Jia","Liheng Chen","Zhoumianze Liu","Ben Kao","Guohao Li","Junxian He","Yu Qiao","Zhiyong Wu"],"authors_zh":"Qiushi Sun、Kanzhi Cheng、Zichen Ding、Chuanyang Jin、Yian Wang、Fangzhi Xu、Zhenyu Wu、Chengyou Jia、Liheng Chen、Zhoumianze Liu、Ben Kao、Guohao Li、Junxian He、Yu Qiao、Zhiyong Wu","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","agent_environment","construction_recipe"],"verification_contract":["environmental"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction"],"tags":["environment-agent-trajectory-data","agent-trajectories","evaluation"],"status":"partial","priority":"必读","paper_type_zh":"GUI 智能体轨迹构造方法与数据发布","best_for_zh":"研究环境交互轨迹、反向任务合成、轨迹奖励与 GUI 智能体训练的读者","confidence":"medium","one_line":["OS-Genesis explores virtual GUIs first, reverse-synthesizes executable tasks from state-action-state triples, and uses graded trajectory rewards to sample mobile and web agent trajectories for SFT.","OS-Genesis 先探索虚拟 GUI，再从状态—动作—状态三元组反向合成可执行任务，并用分级轨迹奖励采样移动端与网页智能体轨迹用于 SFT。"],"why":"It reverses instruction-first collection and preserves incomplete but potentially useful interaction evidence through reward-weighted sampling instead of binary filtering.","primary_link":"https://aclanthology.org/2025.acl-long.277/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OS-Copilot/OS-Genesis"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OS-Copilot/OS-Genesis-mobile-data"},{"key":"project","label":["Project","项目主页"],"url":"https://qiushisun.github.io/OS-Genesis-Home/"}],"link_count":6,"sections":9},{"id":"os-harm-2025","title":"OS-Harm: A Benchmark for Measuring Safety of Computer Use Agents","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track (Spotlight)","authors":["Thomas Kuntz","Agatha Duzan","Hao Zhao","Francesco Croce","Zico Kolter","Nicolas Flammarion","Maksym Andriushchenko"],"authors_zh":"Thomas Kuntz, Agatha Duzan, Hao Zhao, Francesco Croce, Zico Kolter, Nicolas Flammarion, Maksym Andriushchenko","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","audit_failure"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","computer_use","safety"],"tags":["computer-use-agents","osworld","agent-safety","prompt-injection","trajectory-judging","llm-as-judge","benchmark-audit"],"status":"partial","priority":"必读","paper_type_zh":"计算机使用智能体安全基准与轨迹审计研究","best_for_zh":"研究桌面智能体安全评测、轨迹judge、prompt injection与环境回放审计的读者","confidence":"high","one_line":["OS-Harm evaluates computer-use agents on 150 OSWorld tasks spanning misuse, prompt injection, and model misbehavior, with multimodal episode logs and a GPT-4.1 AER judge whose safety/completion F1 against 150 human-labeled o4-mini traces is 0.76/0.79.","OS-Harm以150个OSWorld Ubuntu VM任务评测计算机使用智能体的误用、prompt injection与模型失当，并用GPT-4.1 AER输出安全/完成判定及首个违规步骤，但judge召回不足与50/51条注入配置漂移限制了复现。"],"why":"It makes agent-safety feedback auditable at both episode and first-unsafe-step granularity while showing why semantic judging cannot be treated as ground truth: unsafe recall is 64%, failure modes are trace-format dependent, and the mutable release currently disagrees with the paper's prompt-injection count.","primary_link":"https://arxiv.org/abs/2506.14866","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tml-epfl/os-harm"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/drive/folders/1l_Il_Kcx8XGZvh2RunBX4GlRUDpp2ypn"},{"key":"project","label":["Project","项目主页"],"url":"https://thomaskuntz.org/projects/os-harm/"}],"link_count":7,"sections":9},{"id":"osworld-mcp-2025","title":"OSWorld-MCP: Benchmarking MCP Tool Invocation In Computer-Use Agents","year":2025,"venue":"ICLR 2026","authors":["Hongrui Jia","Jitong Liao","Xi Zhang","Haiyang Xu","Tianbao Xie","Chaoya Jiang","Ming Yan","Si Liu","Wei Ye","Fei Huang"],"authors_zh":"Hongrui Jia, Jitong Liao, Xi Zhang, Haiyang Xu, Tianbao Xie, Chaoya Jiang, Ming Yan, Si Liu, Wei Ye, Fei Huang","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use"],"tags":["agent-environment","agent-trajectories","evaluation","desktop-gui","mcp","tool-use"],"status":"partial","priority":"必读","paper_type_zh":"桌面 GUI/MCP 智能体环境、基准与工具发布","best_for_zh":"研究混合 GUI/工具 episode、环境 verifier、工具选择反馈、轨迹重放与 MCP 安全审计的读者","confidence":"high","one_line":["OSWorld-MCP evaluates 361 OSWorld desktop tasks with 158 curated MCP tools, per-step GUI/tool choice, task-specific environment checks, and tool-choice-aware TIR, while complete paper trajectories and immutable replay artifacts remain unreleased.","OSWorld-MCP 在 361 个 OSWorld 桌面任务上引入 158 个经筛选的 MCP 工具，让智能体逐步选择 GUI 或工具，并用任务特定环境检查与 TIR 评估结果；论文实验的完整轨迹和不可变重放材料尚未发布。"],"why":"It makes hybrid GUI/MCP decision episodes and verifier feedback inspectable, but also exposes split leakage, evaluator scope, mutable tool retrieval, missing rollout retention, unclear licensing, and a broad MCP attack surface that constrain training reuse.","primary_link":"https://arxiv.org/abs/2510.24563","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/X-PLUG/OSWorld-MCP"},{"key":"data","label":["Data","数据"],"url":"https://github.com/X-PLUG/OSWorld-MCP/tree/main/evaluation_examples"},{"key":"project","label":["Project","项目主页"],"url":"https://osworld-mcp.github.io/"}],"link_count":5,"sections":9},{"id":"paperbench-2025","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","year":2025,"venue":"ICML 2025 (PMLR 267)","authors":["Giulio Starace","Oliver Jaffe","Dane Sherburn","James Aung","Jun Shern Chan","Leon Maksin","Rachel Dias","Evan Mays","Benjamin Kinsella","Wyatt Thompson","Johannes Heidecke","Amelia Glaese","Tejal Patwardhan"],"authors_zh":"Giulio Starace, Oliver Jaffe, Dane Sherburn, James Aung, Jun Shern Chan, Leon Maksin, Rachel Dias, Evan Mays, Benjamin Kinsella, Wyatt Thompson, Johannes Heidecke, Amelia Glaese, Tejal Patwardhan","tracks":["environment_agent_trajectory_data","judgment_rubric_domain_expert_data"],"source_role":["benchmark","agent_environment","verifier_reward","data_release"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","step_level","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","machine_learning_research","scientific_agents","code_agents","long_horizon_tasks","benchmark_evaluation"],"tags":["environment-agent-trajectory-data","ai-research-agents","paper-replication","code-agents","long-horizon-tasks","hierarchical-rubric","llm-as-judge","clean-container-reproduction","blacklist-monitor","partial-credit","contamination","trajectory-release-gap","replay-risk","security"],"status":"partial","priority":"必读","paper_type_zh":"长时程科研智能体复现实验基准与分层验证","best_for_zh":"关注科研智能体、代码执行、分层评分、轨迹发布边界、裁判可靠性与可回放审计的读者","confidence":"high","one_line":["PaperBench evaluates 20 from-scratch ML-paper replications with 8,316 weighted rubric leaves, fresh-container execution, and an o3-mini judge, but releases no complete 646-run trajectory/submission corpus and remains exposed to judge, contamination, rights, replay, and secret-handling risks.","PaperBench 以 20 篇机器学习论文、8,316 个加权 rubric 叶节点和新鲜容器复现评估科研智能体，但官方并未发布覆盖 646 次论文实验的完整日志、快照、提交物、执行产物与逐叶裁判输出。"],"why":"It turns an open-ended research-replication episode into auditable Code Development, Execution, and Result Match checkpoints and separates agent work from clean reproduction. That same design shows why public task/rubric packages are evaluation assets, not automatically trajectory-training data: the actual rollouts are absent, result reproduction remains near zero, and judge/replay/ contamination/security boundaries materially affect any claimed capability.","primary_link":"https://proceedings.mlr.press/v267/starace25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/frontier-evals"},{"key":"data","label":["Data","数据"],"url":"https://github.com/openai/frontier-evals/tree/main/project/paperbench/data"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/paperbench/"}],"link_count":11,"sections":9},{"id":"phi-4-mini-reasoning-2025","title":"Phi-4-Mini-Reasoning: Exploring the Limits of Small Reasoning Language Models in Math","year":2025,"venue":"arXiv preprint","authors":["Haoran Xu","Baolin Peng","Hany Awadalla","Dongdong Chen","Yen-Chun Chen","Mei Gao","Young Jin Kim","Yunsheng Li","Liliang Ren","Yelong Shen","Shuohang Wang","Weijian Xu","Jianfeng Gao","Weizhu Chen"],"authors_zh":"Haoran Xu 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["distillation","preference_learning","rlvr","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["mathematics"],"tags":["phi-4-mini","frontier-report","disclosure-ledger","mathematical-reasoning","synthetic-cot","dpo","rlvr"],"status":"partial","priority":"必读","paper_type_zh":"小模型推理技术报告与数据披露账本","best_for_zh":"审计数学推理蒸馏、偏好学习和 RLVR 披露边界的读者","confidence":"high","one_line":["Phi-4-Mini-Reasoning documents a four-stage math-reasoning pipeline over about 10M rollouts, with correct-trace distillation, rejected-rollout DPO, and final-answer RLVR, but releases weights rather than the training data or verifier stack.","Phi-4-Mini-Reasoning 报告了合成 CoT、Rollout DPO 与答案可验证 RL 的四阶段数学推理配方，但未发布完整数据、verifier 或审计资产。"],"why":"It enables a stage-by-stage disclosure audit of data objects and feedback contracts for a 3.8B reasoning model, including a visible mismatch between the paper and later model-card descriptions of trajectory provenance.","primary_link":"https://arxiv.org/abs/2504.21233","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/microsoft/Phi-4-mini-reasoning"},{"key":"project","label":["Project","项目主页"],"url":"https://azure.microsoft.com/en-us/blog/one-year-of-phi-small-language-models-making-big-leaps-in-ai/"}],"link_count":5,"sections":9},{"id":"mixture-of-thoughts-2025","title":"Phi-4-reasoning Technical Report","year":2025,"venue":"arXiv preprint (2025)","authors":["Marah Abdin","Sahaj Agarwal","Ahmed Awadallah","Vidhisha Balachandran","Harkirat Behl","Lingjiao Chen","Gustavo de Rosa","Suriya Gunasekar","Mojan Javaheripi","Neel Joshi","Piero Kauffmann","Arindam Mitra","Besmira Nushi","Dimitris Papailiopoulos","Guoqing Zheng"],"authors_zh":"Marah Abdin、Sahaj Agarwal、Ahmed Awadallah、Vidhisha Balachandran、Harkirat Behl、Lingjiao Chen、Gustavo de Rosa、Suriya Gunasekar、Mojan Javaheripi、Neel Joshi、Piero Kauffmann、Arindam Mitra、Besmira Nushi、Dimitris Papailiopoulos、Guoqing Zheng","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English mathematics, coding, science, planning, and algorithmic reasoning"],"tags":["instruction-demonstration-rationale","arxiv-2504.21318","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"Phi-4-reasoning 的监督微调","confidence":"high","one_line":["Mixture-of-Thoughts selects teachable prompts and packages teacher reasoning from several domains into a source-labeled conversation mixture.","Mixture-of-Thoughts 先挑选适合学生学习的题目，再汇集 34.9 万条跨数学、代码与科学的教师推理对话。"],"why":"Reasoning distillation often maximizes trace volume without selecting prompts at a level where a student can learn, causing expensive demonstrations to add little usable supervision.","primary_link":"https://arxiv.org/abs/2504.21318","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/open-r1/Mixture-of-Thoughts"}],"link_count":2,"sections":9},{"id":"physreason-physics-based-reasoning-2025","title":"PhysReason: A Comprehensive Benchmark towards Physics-Based Reasoning","year":2025,"venue":"ACL 2025","authors":["Xinyu Zhang","Yuxuan Dong","Yanrui Wu","Jiaxing Huang","Chengyou Jia","Basura Fernando","Mike Zheng Shou","Lingling Zhang","Jun Liu"],"authors_zh":"Xinyu Zhang、Yuxuan Dong、Yanrui Wu、Jiaxing Huang、Chengyou Jia、Basura Fernando、Mike Zheng Shou、Lingling Zhang、Jun Liu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["physics-reasoning","multimodal-reasoning","process-evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要评测文本与图示结合的物理推理，并定位具体过程错误类型的研究者。","confidence":"high","one_line":["PhysReason evaluates 1,200 diagram-aware physics problems with answer and step-level scoring, exposing whether errors arise from theorem use, process understanding, calculation, or conditions.","PhysReason 用 1,200 道含图物理题同时评测答案与逐步推理，区分定理应用、过程理解、计算和条件分析四类失误。"],"why":"It extends process-level evaluation beyond mathematics to physics tasks with diagrams, formulas, and condition-sensitive reasoning.","primary_link":"https://aclanthology.org/2025.acl-long.811/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wxxy0719/PhysReason"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zhibei1204/PhysReason"},{"key":"project","label":["Project","项目主页"],"url":"https://dxzxy12138.github.io/PhysReason/"}],"link_count":6,"sections":9},{"id":"pku-saferlhf-multi-level-safety-preference-2025","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","year":2025,"venue":"ACL 2025","authors":["Jiaming Ji","Donghai Hong","Borong Zhang","Boyuan Chen et al."],"authors_zh":"Jiaming Ji、Donghai Hong、Borong Zhang、Boyuan Chen 等","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["preference-feedback-batch-2026","post-training","data-construction"],"status":"verified","priority":"可读","paper_type_zh":"偏好与奖励反馈数据集或数据构建研究","best_for_zh":"需要构建、审计或复用偏好与奖励反馈数据的研究者。","confidence":"high","one_line":["PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference contributes a preference/reward feedback data object or construction method.","发布 PKU-SafeRLHF 安全偏好数据集，分离有用性与无害性偏好，并加入 19 类风险和三级严重度标签。"],"why":"It exposes a reusable preference or reward-feedback data surface that requires provenance and bias audit before reuse.","primary_link":"https://arxiv.org/abs/2406.15513","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PKU-Alignment/safe-rlhf"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/PKU-Alignment/PKU-SafeRLHF"}],"link_count":3,"sections":9},{"id":"planetarium-text-structured-planning-2025","title":"Planetarium: A Rigorous Benchmark for Translating Text to Structured Planning Languages","year":2025,"venue":"NAACL 2025","authors":["Max Zuo","Francisco Piedrahita Velez","Xiaochen Li","Michael L. Littman","Stephen H. Bach"],"authors_zh":"Max Zuo, Francisco Piedrahita Velez, Xiaochen Li, Michael L. Littman, Stephen H. Bach","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["planning","pddl","code-generation"],"tags":["pddl","planning","semantic-equivalence","naacl-2025"],"status":"verified","priority":"可读","paper_type_zh":"语义验证的 text-to-PDDL 基准","best_for_zh":"研究文本规划、PDDL 生成或可验证奖励的研究者。","confidence":"high","one_line":["Planetarium pairs four textual descriptions with planning problems and uses object-renaming-aware PDDL equivalence to distinguish solvable outputs from semantically correct ones.","Planetarium 以对象重命名不变的 PDDL 语义等价检查，区分可求解的规划输出与真正表达同一初始状态和目标的输出。"],"why":"It shows that executable plans can encode the wrong problem and makes semantic verification directly reusable for text-to-PDDL training and evaluation.","primary_link":"https://aclanthology.org/2025.naacl-long.560/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/BatsResearch/planetarium"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/BatsResearch/planetarium"}],"link_count":4,"sections":9},{"id":"policy-guided-tree-search-2025","title":"Policy Guided Tree Search for Enhanced LLM Reasoning","year":2025,"venue":"ICML 2025","authors":["Yang Li"],"authors_zh":"Yang Li","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","scalar_reward","trajectory_value"],"training_use":["agent_training","evaluation","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["mathematics","commonsense_reasoning","logical_reasoning","planning"],"tags":["policy-guided-tree-search","learned-search-policy","tree-search","backtracking","ppo","graph-transformer","state-action-traces","scalar-reward","test-time-compute","raw-trees-not-released","paper-code-drift"],"status":"partial","priority":"必读","paper_type_zh":"学习式树搜索控制器、状态—动作轨迹配方与测试时计算研究","best_for_zh":"研究推理树导航策略、标量反馈、分支与回溯记录、测试时预算归因及搜索数据发布审计的读者","confidence":"high","one_line":["PGTS trains a GPS graph policy with PPO to navigate LLaMA 3.1 reasoning trees through expand, branch, backtrack, and paper-defined terminate decisions under explicit depth, breadth, and search-step budgets.","PGTS 用 PPO 训练 GPS 图策略，在显式宽度、深度和搜索步数预算下导航 LLaMA 3.1 的部分推理树；代码可表示未访问或已放弃分支，但未发布论文运行的原始树、完整策略分布、checkpoint 或逐样本预算日志。"],"why":"It exposes the learned controller's state-action-reward contract and the compute tradeoff behind test-time tree search, while the absent raw trees, unchosen-action scores, checkpoints, and explicit code-level terminate action identify the exact evidence still needed for reuse and audit.","primary_link":"https://proceedings.mlr.press/v267/li25bv.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/leao1995/llm_reasoning"},{"key":"data","label":["Data","数据"],"url":"https://github.com/leao1995/llm_reasoning/tree/master/data"}],"link_count":7,"sections":9},{"id":"praetor-fine-grained-llm-evaluator-2025","title":"Praetor: A Fine-Grained Generative LLM Evaluator with Instance-Level Customizable Evaluation Criteria","year":2025,"venue":"ACL 2025","authors":["Yongqi Leng","Renren Jin","Yue Chen","Zhuowen Han","Ling Shi","Jianxiang Peng","Lei Yang","Juesi Xiao","Deyi Xiong"],"authors_zh":"Yongqi Leng、Renren Jin、Yue Chen、Zhuowen Han、Ling Shi、Jianxiang Peng、Lei Yang、Juesi Xiao、Deyi Xiong","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["Praetor trains a bilingual judge to apply optional or instance-specific criteria in pointwise and pairwise evaluation.","Praetor 训练可在点式和成对评测中使用可选或实例化标准的中英双语评判模型。"],"why":"Praetor trains a bilingual judge to apply optional or instance-specific criteria in pointwise and pairwise evaluation.","primary_link":"https://aclanthology.org/2025.acl-long.513/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tjunlp-lab/Praetor"}],"link_count":3,"sections":9},{"id":"prolog-math-2025","title":"Predicate-Guided Generation for Mathematical Reasoning","year":2025,"venue":"EMNLP 2025","authors":["Jiajun Chen","Yik-Cheung Tam"],"authors_zh":"Jiajun Chen、Yik-Cheung Tam","tracks":["data_construction_open_release_recipes","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematical_reasoning","logic_programming"],"tags":["prolog-math","predicate-guided-generation","swi-prolog","math-verify","symbolic-equivalence","grpo"],"status":"partial","priority":"必读","paper_type_zh":"可执行符号推理数据发布与构造配方","best_for_zh":"研究数学数据构造、程序执行验证、predicate 复用与 RLVR 反馈边界的读者","confidence":"high","one_line":["Prolog-MATH translates 7,500 MATH training problems and CoTs into predicate-guided SWI-Prolog candidates, filters terminal answers symbolically, and applies SFT/GRPO recovery; the 97.4% paper union is not present as a reconciled public manifest.","Prolog-MATH 将 7,500 道 MATH 训练题及 CoT 转成 predicate-guided SWI-Prolog 候选，以终态符号等价筛选并用 SFT/GRPO 恢复失败；论文所述 97.4% 并集尚无可核对的公开清单。"],"why":"It makes predicate reuse, typed outputs, execution filtering, retries, and program-level RL feedback concrete while exposing the difference between terminal correctness and faithful reasoning.","primary_link":"https://aclanthology.org/2025.emnlp-main.462/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tinyyhope/Prolog-MATH"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Tinyhope/Prolog-MATH"}],"link_count":5,"sections":9},{"id":"preference-leakage-2025","title":"Preference Leakage: A Contamination Problem in LLM-as-a-Judge","year":2025,"venue":"ICML 2025","authors":["Dawei Li","Renliang Sun","Yue Huang","Ming Zhong","Bohan Jiang","Jiawei Han","Xiangliang Zhang","Wei Wang","Huan Liu"],"authors_zh":"Dawei Li, Renliang Sun, Yue Huang, Ming Zhong, Bohan Jiang, Jiawei Han, Xiangliang Zhang, Wei Wang, Huan Liu","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","candidate-slate"],"status":"verified","priority":"可读","paper_type_zh":"LLM-as-a-judge 的合成数据污染、谱系偏差与评测审计研究","best_for_zh":"使用合成偏好数据和 LLM judge、需审计模型谱系偏差的研究者。","confidence":"high","one_line":["judge relatedness bias and released synthesis/judgment data","揭示生成器与 judge 的谱系关联会使 judge 偏向相关学生模型。"],"why":"It offers a concrete audit surface or failure-mode dataset for Track 13.","primary_link":"https://openreview.net/forum?id=grIvSXVJ65","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/David-Li0406/Preference-Leakage"}],"link_count":3,"sections":9},{"id":"pfpo-pseudo-feedback-reasoning-2025","title":"Preference Optimization for Reasoning with Pseudo Feedback","year":2025,"venue":"ICLR 2025 Spotlight","authors":["Fangkai Jiao","Geyang Guo","Xingxing Zhang","Nancy F. Chen","Shafiq Joty","Furu Wei"],"authors_zh":"Fangkai Jiao、Geyang Guo、Xingxing Zhang、Nancy F. Chen、Shafiq Joty、Furu Wei（南洋理工大学、微软研究院、A*STAR I2R、佐治亚理工学院、Salesforce Research）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["preference_learning"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning","code-generation"],"tags":["pseudo-feedback","preference-data","test-cases","dpo"],"status":"verified","priority":"必读","paper_type_zh":"伪反馈偏好数据构造与推理优化研究","best_for_zh":"适合缺少专家标注、但拥有可执行或可核对答案测试的推理任务开发者。","confidence":"high","one_line":["PFPO constructs preference pairs for math and code reasoning by evaluating sampled solutions against frontier-generated or self-consistent pseudo test cases.","PFPO 通过在前沿模型生成或自一致伪测试用例上评估采样解答，为数学与代码推理构造偏好回答对。"],"why":"It makes the test case, pass score, and pair-margin rule explicit training objects rather than assuming a human preference label already exists.","primary_link":"https://arxiv.org/abs/2411.16345","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/unilm/tree/master/PFPO"}],"link_count":3,"sections":9},{"id":"selective-dpo-difficulty-selection-2025","title":"Principled Data Selection for Alignment: The Hidden Risks of Difficult Examples","year":2025,"venue":"arXiv preprint","authors":["Chengqian Gao","Haonan Li","Liu Liu","Zeke Xie","Peilin Zhao","Zhiqiang Xu"],"authors_zh":"Chengqian Gao、Haonan Li、Liu Liu、Zeke Xie、Peilin Zhao、Zhiqiang Xu","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-following"],"tags":["data-selection","dpo","preference-learning","curriculum","difficulty"],"status":"verified","priority":"必读","paper_type_zh":"偏好样本难度选择与直接偏好优化研究","best_for_zh":"适合研究模型容量如何决定偏好记录可训练性的读者。","confidence":"high","one_line":["Selective DPO filters preference pairs that are too difficult for the current model capacity.","Selective DPO 按模型容量筛除过难的偏好对，再执行直接偏好优化。"],"why":"It treats data difficulty as a training-use decision rather than assuming every clean preference pair is beneficial.","primary_link":"https://arxiv.org/abs/2502.09650","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/glorgao/SelectiveDPO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuggingFaceH4/ultrafeedback_binarized"}],"link_count":4,"sections":9},{"id":"prmbench-fine-grained-process-reward-benchmark-2025","title":"PRMBENCH: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","year":2025,"venue":"ACL 2025","authors":["Mingyang Song","Zhaochen Su","Xiaoye Qu","Jiawei Zhou","Yu Cheng"],"authors_zh":"Mingyang Song, Zhaochen Su, Xiaoye Qu, Jiawei Zhou, Yu Cheng","tracks":["preference_reward_feedback_data","process_trace_supervision_data","audit_failure_contamination_verifier_attacks"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-reward-modeling","evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"细粒度过程奖励模型评测与步骤反馈基准论文","best_for_zh":"需要评测或诊断过程奖励模型对数学推理中隐性错误、合理性和敏感性的识别能力的研究者。","confidence":"high","one_line":["PRMBENCH provides 6,216 mathematical problems and 83,456 step labels to measure whether process reward models detect subtle reasoning errors.","6,216题、83,456步标签，以正确性/合理性等细粒度反馈训练和诊断 PRM，直接服务多步数学推理。"],"why":"Its released error-step and reason fields let builders audit a PRM beyond final-answer correctness.","primary_link":"https://aclanthology.org/2025.acl-long.1230/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ssmisya/PRMBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hitsmy/PRMBench_Preview"},{"key":"project","label":["Project","项目主页"],"url":"https://prmbench.github.io/"}],"link_count":6,"sections":9},{"id":"thinkprm-2025","title":"Process Reward Models That Think","year":2025,"venue":"Transactions on Machine Learning Research (TMLR), 2026","authors":["Muhammad Khalifa","Rishabh Agarwal","Lajanugen Logeswaran","Jaekyeom Kim","Hao Peng","Moontae Lee","Honglak Lee","Lu Wang"],"authors_zh":"unknown","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["verifier_reward","process_supervision","construction_recipe","data_release","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","full_episode","scalar_reward","process_reward"],"training_use":["sft","reward_modeling","process_supervision","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report","release_audit"],"domains":["math","science","code"],"tags":["thinkprm","generative-prm","verification-cot","process-supervision","rejection-sampling","best-of-n","verifier-guided-search","test-time-verifier-scaling","open-release"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["ThinkPRM trains generative process verifiers on 1,000 QwQ-generated, PRM800K-label-filtered verification CoTs and spends additional verifier compute to score or select reasoning paths.","ThinkPRM 生成逐步验证推理，并把验证计算用于候选选择和奖励引导搜索。"],"why":"It makes verifier reasoning a first-class test-time trace and shows how process-label filtering, score extraction, repeated verification, and search aggregation jointly determine candidate selection, while leaving rationale validity and release lineage as separate audit questions.","primary_link":"https://arxiv.org/abs/2504.16828","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mukhal/thinkprm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/launch/thinkprm-1K-verification-cots"},{"key":"project","label":["Project","项目主页"],"url":"https://mukhal.github.io/thinkprm/"}],"link_count":6,"sections":9},{"id":"process-vs-outcome-reward-agentic-rag-reinforcement-learning-2025","title":"Process vs. Outcome Reward: Which is Better for Agentic RAG Reinforcement Learning","year":2025,"venue":"NeurIPS 2025","authors":["Wenlin Zhang","Xiangyang Li","Kuicai Dong","Yichao Wang","Pengyue Jia","Xiaopeng Li","Yingyi Zhang","Derong Xu","Zhaocheng Du","Huifeng Guo","Ruiming Tang","Xiangyu Zhao"],"authors_zh":"Wenlin Zhang、Xiangyang Li、Kuicai Dong 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agentic_rag","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"智能检索增强生成的过程奖励、偏好优化与监督数据论文","best_for_zh":"训练或评测查询生成、证据抽取和答案生成的过程级检索策略。","confidence":"high","one_line":["ReasonRAG releases RAG-ProGuide, 13K process-level preference pairs from MCTS and shortest-path reward estimation for training agentic retrieval policies.","ReasonRAG 发布 RAG-ProGuide：由 MCTS 和最短路径奖励估计构造的 1.3 万个过程级偏好对，用于训练智能检索策略。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/file/54e1381d0c0598127b90af4c940fd3d9-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Applied-Machine-Learning-Lab/ReasonRAG"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/reasonrag/RAG_ProGuide"}],"link_count":4,"sections":9},{"id":"process-based-self-rewarding-language-models-2025","title":"Process-based Self-Rewarding Language Models","year":2025,"venue":"arXiv","authors":["Shimao Zhang","Xiao Liu","Xin Zhang","Junxiao Liu","Zheheng Luo","Shujian Huang","Yeyun Gong"],"authors_zh":"Shimao Zhang、Xiao Liu、Xin Zhang、Junxiao Liu、Zheheng Luo、Shujian Huang、Yeyun Gong","tracks":["process_trace_supervision_data"],"source_role":["process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"可读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["Process-based Self-Rewarding Language Models addresses process-based self-rewarding supervision.","过程式自奖励把长思维、逐步 LLM 裁判和逐步偏好优化引入数学推理的自奖励流程。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2503.03746","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Shimao-Zhang/Process-Self-Rewarding"}],"link_count":2,"sections":9},{"id":"enconda-bench-2025","title":"Process-Level Trajectory Evaluation for Environment Configuration in Software Engineering Agents","year":2025,"venue":"ICLR 2026 (Poster)","authors":["Jiayi Kuang","Yinghui Li","Xin Zhang","Yangning Li","Di Yin","Xing Sun","Ying Shen","Philip S. Yu"],"authors_zh":"Jiayi Kuang, Yinghui Li, Xin Zhang, Yangning Li, Di Yin, Xing Sun, Ying Shen, Philip S. Yu","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software_engineering","repository_level_code","code_agents","python","environment_configuration","dependency_management","environment_interaction"],"tags":["environment-agent-trajectory-data","software-engineering-agent","environment-configuration","dependency-repair","synthetic-readme-errors","process-level-evaluation","executable-verifier","llm-judge","missing-trajectories","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"软件环境配置智能体的过程级评测基准与任务数据发布","best_for_zh":"研究环境型软件工程 agent、过程 verifier、可执行终止条件、benchmark replay 与发布审计的读者","confidence":"high","one_line":["EnConda-Bench turns pinned Python repositories and minimally corrupted READMEs into 4,201 environment-repair tasks scored by error labels, LLM-judged fixes, and Docker build/test success; the public release contains task records rather than replayable agent episodes.","EnConda-Bench 将固定版本的 Python 仓库与最小化破坏的 README 组织成 4,201 个环境修复任务，以错误标签、LLM 语义判断和 Docker 构建/测试终止条件评分；公开发布的是任务与金标修复记录，不是可回放的 agent trajectory。"],"why":"It shows how environment-agent evaluation can expose the gap between diagnosing a dependency/setup error and executing a robust repair, while making clear that repository pins, containers, package/network state, terminal predicates, and retained failures all belong to the reasoning-data contract.","primary_link":"https://openreview.net/forum?id=Q8qgloDKUO","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TencentYoutuResearch/EnConda-Bench/tree/86ab7858613b85f4a8316f3cda3c83086b8cf7c2"},{"key":"data","label":["Data","数据"],"url":"https://github.com/TencentYoutuResearch/EnConda-Bench/tree/86ab7858613b85f4a8316f3cda3c83086b8cf7c2/Benchmark_Data"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/TencentYoutuResearch/EnConda-Bench"}],"link_count":6,"sections":9},{"id":"interactive-multimodal-tool-use-rl-2025","title":"Process-Supervised Reinforcement Learning for Interactive Multimodal Tool-Use Agents","year":2025,"venue":"arXiv preprint; submitted to ICLR 2026","authors":["Weiting Tan","Xinghua Qu","Ming Tu","Meng Ge","Andy T. Liu","Philipp Koehn","Lu Lu"],"authors_zh":"Weiting Tan, Xinghua Qu, Ming Tu, Meng Ge, Andy T. Liu, Philipp Koehn, Lu Lu","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","process_supervision","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode","scalar_reward","process_reward"],"training_use":["process_supervision","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","environment_interaction","tool_use","multimodal_speech","mathematics"],"tags":["environment-agent-trajectory-data","process-supervision","turn-level-reward","tarl","agentic-rl","multimodal-tool-use","speech-text-rollouts"],"status":"partial","priority":"可读","paper_type_zh":"交互式多模态工具使用强化学习与轨迹奖励研究","best_for_zh":"研究智能体强化学习、轨迹奖励、语音工具使用和 verifier 审计的读者","confidence":"high","one_line":["Each 1-30-turn retail sandbox rollout carries interleaved agent/environment tokens, GPT-4.1 turn scores, and a rule-verified terminal outcome, but the promised sandbox, tasks, rollouts, and checkpoints are not linked as released artifacts.","TARL 为约3,000个零售任务的文本与语音工具使用 rollout 组合规则终局奖励和 GPT-4.1 逐轮评分，但稳定训练实际优化聚合后的轨迹标量，且论文承诺的 sandbox、任务、rollout 与 checkpoint 尚无已核验发布。"],"why":"It makes a subtle supervision boundary auditable: fine-grained judge labels are generated per turn, yet the stable optimizer consumes a full-episode scalar. That distinction matters when comparing process supervision, reward shaping, and replayable agent-trajectory data.","primary_link":"https://arxiv.org/abs/2509.14480","links":[],"link_count":3,"sections":9},{"id":"prods-preference-oriented-selection-2025","title":"ProDS: Preference-oriented Data Selection for Instruction Tuning","year":2025,"venue":"arXiv preprint","authors":["Wenya Guo","Zhengkun Zhang","Xumeng Liu","Ying Zhang","Ziyu Lu","Haoze Zhu","Xubo Liu","Ruxue Yan"],"authors_zh":"Wenya Guo、Zhengkun Zhang、Xumeng Liu 等（南开大学、百度）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["instruction-following","alignment","reasoning"],"tags":["data-selection","preference-learning","instruction-tuning","sft"],"status":"verified","priority":"必读","paper_type_zh":"偏好导向的指令数据选择研究","best_for_zh":"适合需要让监督微调数据贴合特定目标任务偏好的读者。","confidence":"high","one_line":["ProDS selects instruction records by their alignment with positive and negative target preference directions.","ProDS 用目标任务的正反偏好方向衡量样本影响，从而选择更适合监督微调的指令记录。"],"why":"It treats preference alignment as a criterion for SFT-data selection rather than only a final optimization objective.","primary_link":"https://arxiv.org/abs/2505.12754","links":[],"link_count":2,"sections":9},{"id":"profbench-professional-rubrics-2025","title":"ProfBench: Multi-Domain Rubrics requiring Professional Knowledge to Answer and Judge","year":2025,"venue":"ICLR 2026","authors":["Zhilin Wang","Jaehun Jung","Ximing Lu","Shizhe Diao","Ellie Evans","Jiaqi Zeng","Pavlo Molchanov","Yejin Choi","Jan Kautz","Yi Dong"],"authors_zh":"Zhilin Wang, Jaehun Jung, Ximing Lu, Shizhe Diao, Ellie Evans, Jiaqi Zeng, Pavlo Molchanov, Yejin Choi, Jan Kautz, Yi Dong","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["professional_reasoning","science","finance","consulting"],"tags":["rubric","professional_reasoning","llm_judge","reward_model"],"status":"verified","priority":"必读","paper_type_zh":"多专业领域的回答与 Judge rubric 基准","best_for_zh":"需要训练或比较专业报告 Judge、奖励模型和长文本回答系统的研究者。","confidence":"high","one_line":["ProfBench pairs professional prompts and grounding documents with expert response-rubric supervision across four domains.","ProfBench 为科学、金融和咨询报告提供专家撰写的细粒度准则，可同时评估回答模型与 Judge。"],"why":"It extends rubric-based evaluation beyond domains with easily verified answers.","primary_link":"https://arxiv.org/abs/2510.18941","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVlabs/ProfBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/ProfBench"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/nvidia/ProfBench"}],"link_count":6,"sections":9},{"id":"projudge-multimodal-process-judges-2025","title":"ProJudge: A Multi-Modal Multi-Discipline Benchmark and Instruction-Tuning Dataset for MLLM-based Process Judges","year":2025,"venue":"ICCV 2025","authors":["Jiaxin Ai","Pengfei Zhou","Zhaopan Xu","Ming Li","Fanrui Zhang","Zizhen Li","Jianwen Sun","Yukang Feng","Baojin Huang","Zhongyuan Wang","Kaipeng Zhang"],"authors_zh":"Jiaxin Ai, Pengfei Zhou, Zhaopan Xu, Ming Li, Fanrui Zhang, Zizhen Li, Jianwen Sun, Yukang Feng, Baojin Huang, Zhongyuan Wang, Kaipeng Zhang","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","mathematical-reasoning","scientific-reasoning","process-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要训练或评测理化生和数学场景中多模态过程裁判的研究者。","confidence":"high","one_line":["ProJudge combines a 2,400-case human-labeled multimodal process-judge benchmark with a 173K instruction-tuning release for step-level error diagnosis.","ProJudge 以 2,400 个专家标注样例评测多模态过程裁判，并发布 17.3 万条训练记录强化逐步错误诊断。"],"why":"It separates a human-labeled multimodal evaluation surface from a large instruction-tuning resource for process judges.","primary_link":"https://openaccess.thecvf.com/content/ICCV2025/html/Ai_ProJudge_A_Multi-Modal_Multi-Discipline_Benchmark_and_Instruction-Tuning_Dataset_for_MLLM-based_Process_Judges_ICCV_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JulyAI/ProJudge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/julyai/ProJudge-173k"},{"key":"project","label":["Project","项目主页"],"url":"https://projudge.github.io/"}],"link_count":5,"sections":9},{"id":"pae-autonomous-skill-discovery-internet-agents-2025","title":"Proposer-Agent-Evaluator (PAE): Autonomous Skill Discovery For Foundation Model Internet Agents","year":2025,"venue":"ICML 2025","authors":["Yifei Zhou","Qianlan Yang","Kaixiang Lin","Min Bai","Xiong Zhou","Yu-Xiong Wang","Sergey Levine","Li Erran Li"],"authors_zh":"Yifei Zhou, Qianlan Yang, Kaixiang Lin, Min Bai, Xiong Zhou, Yu-Xiong Wang, Sergey Levine, Li Erran Li","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["web-agents","reinforcement-learning","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"互联网智能体的自主任务发现、结果奖励与网页操作轨迹论文","best_for_zh":"需要研究自主任务提案、视觉结果评价或网页智能体强化学习过程数据的研究者。","confidence":"high","one_line":["PAE releases web-agent tasks and trajectories whose outcome rewards come from an autonomous VLM evaluator after context-aware task proposal.","PAE 让提案器自动发现网页任务、让智能体执行，并以视觉结果评估器产生 RL 奖励，公开任务与示例轨迹。"],"why":"It connects task discovery, real web interaction, and outcome feedback in an openly released training workflow.","primary_link":"https://proceedings.mlr.press/v267/zhou25ah.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/amazon-science/PAE"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yifeizhou/pae-data"},{"key":"project","label":["Project","项目主页"],"url":"https://yanqval.github.io/PAE/"}],"link_count":6,"sections":9},{"id":"provable-tts-scaling-laws-2025","title":"Provable Scaling Laws for the Test-Time Compute of Large Language Models","year":2025,"venue":"NeurIPS 2025","authors":["Yanxi Chen","Xuchen Pan","Yaliang Li","Bolin Ding","Jingren Zhou"],"authors_zh":"Yanxi Chen、Xuchen Pan、Yaliang Li、Bolin Ding、Jingren Zhou（机构：Alibaba Group）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-compute","scaling-laws","candidate-aggregation","pairwise-comparison","theory"],"status":"verified","priority":"必读","paper_type_zh":"测试时计算扩展理论与实证研究（NeurIPS 2025）","best_for_zh":"设计候选生成、比较和聚合流程而不依赖外部验证器的读者。","confidence":"high","one_line":["The paper gives knockout and league aggregation algorithms with provable error decay as black-box LLM inference compute grows.","该论文以淘汰赛和联赛式候选聚合证明：黑箱语言模型的测试时调用增加可使错误率按可刻画规律下降。"],"why":"It states the assumptions under which more generation and comparison calls can reliably improve a final answer.","primary_link":"https://arxiv.org/abs/2411.19477","links":[],"link_count":3,"sections":9},{"id":"proving-coding-interview-fvapps-2025","title":"Proving the Coding Interview: A Benchmark for Formally Verified Code Generation","year":2025,"venue":"LLM4Code at ICSE 2025","authors":["Quinn Dougherty","Ronak Mehta"],"authors_zh":"Quinn Dougherty, Ronak Mehta","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","formal-mathematics","lean4"],"tags":["lean4","formal-verification","code-generation","program-synthesis","2025"],"status":"verified","priority":"可读","paper_type_zh":"形式化验证代码生成基准与 Lean 4 数据集","best_for_zh":"需要联合评测程序实现、形式规格与 Lean 证明生成的研究者。","confidence":"high","one_line":["FVAPPS reformulates coding-interview problems as Lean 4 programming-and-proof tasks, with compilation and theorem checking as the final verifier.","FVAPPS 将编程面试题改写为 Lean 4 的程序与证明任务，以编译和定理检查共同验证最终结果。"],"why":"It requires an agent to satisfy a machine-checked specification rather than only pass conventional unit tests.","primary_link":"https://arxiv.org/abs/2502.05714","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/quinn-dougherty/fvapps"}],"link_count":3,"sections":9},{"id":"qcoder-quantum-hardware-feedback-2025","title":"QCoder Benchmark: Bridging Language Generation and Quantum Hardware through Simulator-Based Feedback","year":2025,"venue":"INLG 2025","authors":["Taku Mikuriya","Tatsuya Ishigaki","Masayuki Kawarada","Shunya Minami","Tadashi Kadowaki","Yohichi Suzuki","Soshun Naito","Shunya Takata","Takumi Kato","Tamotsu Basseda","Reo Yamada","Hiroya Takamura"],"authors_zh":"Taku Mikuriya、Tatsuya Ishigaki、Masayuki Kawarada、Shunya Minami、Tadashi Kadowaki、Yohichi Suzuki、Soshun Naito、Shunya Takata、Takumi Kato、Tamotsu Basseda、Reo Yamada、Hiroya Takamura","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["environmental","judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["evaluation","test_time_compute","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["quantum_programming","code_generation","simulator"],"tags":["quantum","simulator_feedback","expert_data","code"],"status":"verified","priority":"可读","paper_type_zh":"领域专家代码数据与环境反馈基准论文","best_for_zh":"研究代码生成、量子计算或 simulator-based feedback 的研究者。","confidence":"high","one_line":["QCoder combines human quantum-programming contest submissions with simulator feedback so LLMs can be judged beyond ordinary Python execution.","QCoder 将量子编程竞赛中的人工代码与量子模拟器反馈结合，评测并改进受硬件约束的代码推理。"],"why":"It demonstrates a domain-expert feedback contract in which executable code must satisfy meaningful hardware-aware constraints.","primary_link":"https://aclanthology.org/2025.inlg-main.43/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QCoder-Bench/QCoder-Benchmark"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tm0012-QCB/QCoderBenchmark"},{"key":"project","label":["Project","项目主页"],"url":"https://qcoder-bench.github.io/"}],"link_count":4,"sections":9},{"id":"questbench-information-acquisition-2025","title":"QuestBench: Can LLMs ask the right question to acquire information in reasoning tasks?","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Belinda Z. Li","Been Kim","Zi Wang"],"authors_zh":"Belinda Z. Li, Been Kim, Zi Wang","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning","planning","mathematics","information-acquisition"],"tags":["clarification","information-acquisition","csp","benchmark","neurips-2025"],"status":"verified","priority":"可读","paper_type_zh":"可验证的主动信息获取推理基准","best_for_zh":"研究澄清提问、主动推理或信息获取策略的研究者。","confidence":"high","one_line":["QuestBench turns clarification into a solver-checked choice: select the single missing variable whose answer makes an underspecified reasoning problem uniquely solvable.","QuestBench 将澄清提问变为可由求解器验证的选择：找出唯一能消除题目剩余解歧义的缺失变量。"],"why":"It separates asking for necessary information from solving an already complete task.","primary_link":"https://arxiv.org/abs/2503.22674","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/belindazli/QuestBench"}],"link_count":3,"sections":9},{"id":"qwen2-5-math-prm-2025","title":"Qwen2.5-Math-PRM","year":2025,"venue":"Qwen release","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward","scalar_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["mathematics","reasoning"],"tags":["qwen","qwen2-5-math-prm","process-reward-model","process-supervision","mathematical-reasoning","verifier-reward","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"数学过程奖励模型发布与验证器披露账本","best_for_zh":"需要区分可用 PRM 推理接口、过程监督证据和未公开训练配方的读者","confidence":"high","one_line":["Qwen released 7B and 72B mathematical process reward models that score segmented reasoning steps through a 0–1 positive-class probability interface, but not their training data, labels, calibration, or construction/audit records.","Qwen 发布了 7B 和 72B 数学过程奖励模型：通过分段推理文本中 extra_0 分隔 token 的正类概率给出 0–1 步级奖励；但训练数据、标签、校准和构造/审计记录没有公开。"],"why":"It separates a usable open PRM interface from the unreleased data and verifier evidence needed to reproduce, calibrate, or independently audit process supervision.","primary_link":"https://arxiv.org/abs/2501.07301","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen2.5-Math-PRM-7B"}],"link_count":5,"sections":9},{"id":"qwen2-5-omni-2025","title":"Qwen2.5-Omni Technical Report","year":2025,"venue":"arXiv preprint","authors":["Jin Xu","Zhifang Guo","Jinzheng He","Hangrui Hu","Ting He","Shuai Bai","Keqin Chen","Jialin Wang","Yang Fan","Kai Dang","Bin Zhang","Xiong Wang","Yunfei Chu","Junyang Lin"],"authors_zh":"Jin Xu 等 14 位作者","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["multimodal","vision","audio","video","speech","text","instruction_following"],"tags":["qwen2-5-omni","frontier-report","data-disclosure-ledger","multimodal","speech","dpo","wer","punctuation-pause-error","instruction-tuning"],"status":"partial","priority":"可读","paper_type_zh":"前沿多模态模型技术报告与数据披露台账","best_for_zh":"需要核查多模态预训练、语音偏好优化和数据披露边界的读者","confidence":"medium","one_line":["Qwen2.5-Omni discloses multimodal pre-training categories and Phase-2 token totals, ChatML instruction-data modalities, and a WER/punctuation-ranked Talker DPO interface, but not the underlying records, reward implementation, sources, or audit artifacts.","Qwen2.5-Omni 披露了多模态预训练类别及第二阶段 token 总量、ChatML 指令数据的模态构成，以及以 WER/标点停顿误差排序的 Talker DPO 接口；但未公开底层记录、奖励实现、数据来源或审计工件。"],"why":"It makes a useful Track 12 disclosure comparison: a frontier release can identify modality mixtures and a pairwise speech-feedback interface while still withholding the source, construction, and validation evidence needed to reproduce or audit it.","primary_link":"https://arxiv.org/abs/2503.20215","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen2.5-Omni"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen2.5-Omni-7B"},{"key":"project","label":["Project","项目主页"],"url":"https://qwenlm.github.io/blog/qwen2.5-omni/"}],"link_count":5,"sections":9},{"id":"qwen2-5-vl-2025","title":"Qwen2.5-VL Technical Report","year":2025,"venue":"arXiv preprint","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","full_episode"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["multimodal","visual_reasoning","document_understanding","OCR","video","agent","mathematics","code"],"tags":["qwen2-5-vl","frontier-report","data-disclosure-ledger","multimodal","synthetic-data","sft","dpo","rejection-sampling","agent-trajectories"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态模型技术报告与数据披露台账","best_for_zh":"需要审计多模态推理、视觉 agent 训练对象、SFT/DPO 与数据披露边界的读者","confidence":"medium","one_line":["Qwen2.5-VL partially discloses 4.1T-scale multimodal training, a roughly 2M-entry SFT mixture, filtering, rejection-sampled reasoning, and SFT/DPO, while withholding reusable data, reward, lineage, and audit artifacts.","Qwen2.5-VL 部分披露了 4.1T 规模多模态训练、约 200 万条 SFT 混合数据、过滤、拒绝采样推理与 SFT/DPO，但未公开可复用数据、reward、lineage 或审计工件。"],"why":"It is a concrete frontier disclosure case: source categories and pipeline interfaces are reported, but enough prompt-, filter-, reward-, and provenance-level detail is absent that the model release and agent capability cannot be treated as a reproducible reasoning-data recipe.","primary_link":"https://arxiv.org/abs/2502.13923","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen2.5-VL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct"},{"key":"project","label":["Project","项目主页"],"url":"https://qwenlm.github.io/blog/qwen2.5-vl/"}],"link_count":4,"sections":9},{"id":"qwen3-2025","title":"Qwen3 Technical Report","year":2025,"venue":"arXiv preprint","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","distillation","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["mathematics","code","reasoning","stem","multilingual","general"],"tags":["qwen3","frontier-report","data-disclosure-ledger","long-cot","grpo","query-verifier-pairs","sft","distillation"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型技术报告与数据披露台账","best_for_zh":"需要核查前沿推理模型训练数据披露、Long-CoT、RL verifier 与可复现性缺口的读者","confidence":"medium","one_line":["Qwen3 partially discloses 36T pretraining, long-CoT filtering, 3,995 GRPO query-verifier pairs, fusion SFT, and distillation while withholding underlying data and verifier/audit details.","Qwen3 技术报告部分披露了 36T 预训练、Long-CoT 过滤、3,995 个 GRPO query-verifier pairs、融合 SFT 与蒸馏，但未公开底层数据、verifier 细节或审计日志。"],"why":"It exposes useful pipeline interfaces and explicit unknowns without making thinking-budget or scaling claims a reproducible recipe.","primary_link":"https://arxiv.org/abs/2505.09388","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen3"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen3-235B-A22B"},{"key":"project","label":["Project","项目主页"],"url":"https://qwenlm.github.io/blog/qwen3/"}],"link_count":5,"sections":9},{"id":"qwen3-coder-2025","title":"Qwen3-Coder: Agentic Coding in the World","year":2025,"venue":"Qwen official release blog","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["coding","software_engineering","agentic_tool_use"],"tags":["qwen3-coder","frontier-report","data-disclosure-ledger","coding-agent","execution-feedback","long-horizon-rl","open-weights"],"status":"partial","priority":"必读","paper_type_zh":"前沿编码智能体发布报告与数据披露账本","best_for_zh":"关注编码 RL、长程智能体环境、执行反馈和开放权重复现边界的研究者","confidence":"high","one_line":["Qwen3-Coder reports 7.5T/70-percent-code pretraining, data cleaning, execution-driven Code RL, and 20,000 parallel Agent-RL environments but not the tasks, tests, trajectories, rewards, or audit records.","Qwen3-Coder 报告了 7.5T、70% code 的预训练、数据清理、execution-driven Code RL 和 20,000 个并行 Agent-RL 环境，但未发布任务、测试、轨迹、reward 或审计记录。"],"why":"It separates released Qwen weights and tooling from the undisclosed post-training substrate behind agentic-coding claims.","primary_link":"https://qwenlm.github.io/blog/qwen3-coder/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen3-Coder"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct"}],"link_count":3,"sections":9},{"id":"qwen3-next-2025","title":"Qwen3-Next: Towards Ultimate Training & Inference Efficiency","year":2025,"venue":"Qwen official release blog","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["scaling_report","frontier_pipeline","release_audit"],"domains":["general_reasoning","long_context","mathematics","code"],"tags":["qwen3-next","qwen","frontier-report","data-disclosure-ledger","hybrid-attention","sparse-moe","gspo","post-training","open-weights"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型发布报告与数据披露账本","best_for_zh":"审计开放权重模型中架构和权重披露与后训练数据、反馈和复现证据之间的边界","confidence":"medium","one_line":["Qwen3-Next releases Instruct and Thinking weights and reports 15T pretraining plus GSPO-assisted RL, but does not disclose post-training prompts, rewards or verifiers, rollout groups, filters, or audit records.","Qwen3-Next 发布了 Instruct 与 Thinking 权重，并披露了 15T 预训练及以 GSPO 改善 RL 稳定性的高层信息；但后训练数据、奖励/验证器、rollout、过滤和审计记录均未公开。"],"why":"It is a high-impact disclosure case in which architecture, inference code, and model weights are inspectable while the data and feedback required to reproduce or audit reasoning post-training remain unavailable.","primary_link":"https://qwen.ai/blog?id=qwen3-next","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking"}],"link_count":2,"sections":9},{"id":"qwen3-vl-2025","title":"Qwen3-VL Technical Report","year":2025,"venue":"arXiv preprint","authors":["Shuai Bai","Yuxuan Cai","Ruizhe Chen","Keqin Chen","Xionghui Chen","Zesen Cheng","Lianghao Deng","Wei Ding","Chang Gao","Chunjiang Ge","Wenbin Ge","Zhifang Guo","Qidong Huang","Jie Huang","Fei Huang","Binyuan Hui","Shutong Jiang","Zhaohai Li","Mingsheng Li","Mei Li","Kaixin Li","Zicheng Lin","Junyang Lin","Xuejing Liu","Jiawei Liu","Chenglong Liu","Yang Liu","Dayiheng Liu","Shixuan Liu","Dunjie Lu","Ruilin Luo","Chenxu Lv","Rui Men","Lingchen Meng","Xuancheng Ren","Xingzhang Ren","Sibo Song","Yuchong Sun","Jun Tang","Jianhong Tu","Jianqiang Wan","Peng Wang","Pengfei Wang","Qiuyue Wang","Yuxuan Wang","Tianbao Xie","Yiheng Xu","Haiyang Xu","Jin Xu","Zhibo Yang","Mingkun Yang","Jianxin Yang","An Yang","Bowen Yu","Fei Zhang","Hang Zhang","Xi Zhang","Bo Zheng","Humen Zhong","Jingren Zhou","Fan Zhou","Jing Zhou","Yuanzhi Zhu","Ke Zhu"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["programmatic","mixed","judgment_required"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","distillation","rlvr","preference_learning","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["multimodal","visual_reasoning"],"tags":["frontier-report","data-disclosure-ledger","qwen","multimodal","long-context","open-code","open-checkpoint"],"status":"partial","priority":"可读","paper_type_zh":"前沿多模态模型技术报告与数据披露台账","best_for_zh":"区分上下文与能力主张、后训练数据披露不足的读者。","confidence":"medium","one_line":["Qwen3-VL discloses a 1.2M-sample multimodal SFT mixture and roughly 30K-query, 16-rollout Reasoning-RL recipe with separate rule and judge rewards, but releases no post-training corpus, rollout ledger, reward configuration, or item-level checkpoint lineage.","Qwen3-VL 报告原生 256K 多模态上下文和图像、视频、文本推理能力，但未披露后训练数据、反馈契约或审计工件。"],"why":"As a frontier disclosure, it distinguishes standard/CoT SFT, long-context curriculum, teacher-response and logit distillation, programmatically verified reasoning RL, and judge-based General RL—while keeping hidden sources and feedback services visibly unresolved.","primary_link":"https://arxiv.org/abs/2511.21631","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen3-VL"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/Qwen/qwen3-vl"}],"link_count":3,"sections":9},{"id":"qwen3guard-2025","title":"Qwen3Guard Technical Report","year":2025,"venue":"arXiv preprint","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","scalar_reward"],"training_use":["sft","safety_alignment","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["safety","content_moderation","reasoning","multilingual"],"tags":["qwen","qwen3guard","frontier-report","data-disclosure-ledger","safety-classification","guardrail","token-level-classification","sft","rlaif"],"status":"partial","priority":"可读","paper_type_zh":"安全护栏技术报告与开源模型发布","best_for_zh":"审计安全分类数据、护栏反馈与流式干预披露边界的读者","confidence":"high","one_line":["Qwen3Guard discloses a 1.19M-scale human/synthetic safety-classification mixture, Gen SFT and Stream token-label construction, and a separate safety-RL use case, while withholding record provenance, full label/reward calibration, and reproducible training artifacts.","Qwen3Guard 披露了约 119 万条人工/合成安全分类样本、Gen 的 SFT、Stream 的 token 级标签构造及一个独立的安全 RL 应用，但未公开记录级来源、完整标签/奖励校准或可复现训练工件。"],"why":"It lets the atlas distinguish a released guard model and partly described safety-data/feedback pipeline from a reusable, fully audited filtering or RL-reward recipe.","primary_link":"https://arxiv.org/abs/2510.14276","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/Qwen3Guard"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Qwen/Qwen3GuardTest"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/Qwen/qwen3guard"},{"key":"project","label":["Project","项目主页"],"url":"https://qwenlm.github.io/blog/qwen3guard/"}],"link_count":7,"sections":9},{"id":"qwq-32b-2025","title":"QwQ-32B: Embracing the Power of Reinforcement Learning","year":2025,"venue":"Qwen official blog","authors":["Qwen Team"],"authors_zh":"Qwen Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","rlvr"],"construction_layer":["reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["mathematics","code","reasoning","general","agents"],"tags":["qwq-32b","qwen","frontier-report","data-disclosure-ledger","rlvr","outcome-based-reward","code-execution-verifier"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型发布报告与数据披露台账","best_for_zh":"需要核查前沿推理模型的数据、反馈契约与可复现性缺口的读者","confidence":"medium","one_line":["QwQ-32B discloses cold-start, outcome-based mathematics/code RL and a later general-RL stage, but withholds the training data, verifier implementations, reward calibration, and audit artifacts.","QwQ-32B 披露了 cold-start、基于结果的数学/代码 RL 及后续通用 RL 阶段，但未公开训练数据、verifier 实现、奖励校准或审计工件。"],"why":"It cleanly separates a public model-weight release and high-level RL narrative from the unavailable data lineage, executable feedback stack, and reproducibility evidence required for reuse.","primary_link":"https://qwenlm.github.io/blog/qwq-32b/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/QwQ"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/QwQ-32B"}],"link_count":3,"sections":9},{"id":"r-prm-reasoning-driven-process-reward-modeling-2025","title":"R-PRM: Reasoning-Driven Process Reward Modeling","year":2025,"venue":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","authors":["Shuaijie She","Junxiao Liu","Yifeng Liu","Jiajun Chen","Xin Huang","Shujian Huang"],"authors_zh":"Shuaijie She、Junxiao Liu、Yifeng Liu、Jiajun Chen、Xin Huang、Shujian Huang","tracks":["process_trace_supervision_data","training_usage_optimization_objectives"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["math_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"推理驱动的过程奖励建模与偏好数据论文","best_for_zh":"训练生成式过程评审器，或研究数学推理的步骤诊断。","confidence":"high","one_line":["R-PRM pairs step-level mathematical judgments with generated rationales and preference optimization to make process evaluation more accurate and useful.","R-PRM 将逐步数学判定与生成式理由及偏好优化结合，提高过程评估的准确性和可用性。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://aclanthology.org/2025.emnlp-main.679/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NJUNLP/R-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/kevinpro/R-PRM"}],"link_count":4,"sections":9},{"id":"rip-prompt-preference-filtering-2025","title":"R.I.P.: Better Models by Survival of the Fittest Prompts","year":2025,"venue":"ICML 2025","authors":["Ping Yu","Weizhe Yuan","Olga Golovneva","Tianhao Wu","Sainbayar Sukhbaatar","Jason Weston","Jing Xu"],"authors_zh":"Ping Yu、Weizhe Yuan、Olga Golovneva 等（Meta、纽约大学、加州大学伯克利分校）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing"],"domains":["alignment","instruction-following"],"tags":["prompt-filtering","preference-learning","synthetic-data","dpo"],"status":"verified","priority":"必读","paper_type_zh":"提示完整性筛选与合成偏好数据构造研究","best_for_zh":"适合希望在偏好优化前筛除含糊或低价值提示的读者。","confidence":"high","one_line":["RIP selects instruction prompts by rejected-answer quality and preference reward gap before DPO training.","RIP 通过拒选回答质量和偏好奖励间隔筛选提示，再构造用于 DPO 的记录。"],"why":"It makes prompt integrity a measurable gate for preference-data construction and consumption.","primary_link":"https://arxiv.org/abs/2501.18578","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/facebook/Wildchat-RIP-Filtered-by-8b-Llama"}],"link_count":4,"sections":9},{"id":"r1-onevision-2025","title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","year":2025,"venue":"arXiv preprint","authors":["Yi Yang","Xiaoxuan He","Hongkun Pan","Xiyan Jiang","Yan Deng","Xingtao Yang","Haoyu Lu","Dacheng Yin","Fengyun Rao","Minfeng Zhu","Bo Zhang","Wei Chen"],"authors_zh":"Yi Yang, Xiaoxuan He, Hongkun Pan, Xiyan Jiang, Yan Deng, Xingtao Yang, Haoyu Lu, Dacheng Yin, Fengyun Rao, Minfeng Zhu, Bo Zhang, Wei Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","data_release","model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","visual-reasoning","mathematical-reasoning"],"tags":["arxiv-2503.10615","multimodal-cot","cross-modal-formalization"],"status":"verified","priority":"可读","paper_type_zh":"多模态推理数据构造与后训练论文","best_for_zh":"适合构建视觉思维链语料，或审计文本推理是否仍由图像证据支撑的读者。","confidence":"high","one_line":["R1-Onevision formalizes images for a language reasoner, filters the resulting multimodal CoT, and releases more than 155K demonstrations for visual reasoning SFT.","R1-Onevision 先把图像形式化为文本表示，再生成并过滤多模态思维链，发布超过 155K 条视觉推理示范用于 SFT。"],"why":"It exposes a concrete image-to-description-to-reasoning record rather than treating multimodal reasoning supervision as an opaque final answer.","primary_link":"https://arxiv.org/abs/2503.10615","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Fancy-MLLM/R1-onevision"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Fancy-MLLM/R1-onevision"}],"link_count":3,"sections":9},{"id":"r1-reward-training-multimodal-reward-model-through-stable-reinforcement-learning-2025","title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","year":2025,"venue":"ICLR 2026","authors":["Yi-Fan Zhang","Xingyu Lu","Xiao Hu","Chaoyou Fu","Bin Wen","Tianke Zhang","Changyi Liu","Kaiyu Jiang","Kaibing Chen","Kaiyu Tang","Haojie Ding","Jiankang Chen","Fan Yang","Zhang Zhang","Tingting Gao","Liang Wang"],"authors_zh":"Yi-Fan Zhang、Xingyu Lu、Xiao Hu、Chaoyou Fu、Bin Wen 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"多模态奖励模型强化学习与训练数据论文","best_for_zh":"研究多模态奖励建模、强化学习稳定性或奖励数据复用的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","R1-Reward 以稳定强化学习训练多模态奖励模型，并公开相应的奖励训练数据和代码。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2505.02835","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yfzhang114/r1_reward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yifanzhang114/R1-Reward-RL"}],"link_count":3,"sections":9},{"id":"r2e-gym-procedural-swe-environments-2025","title":"R2E-Gym: Procedural Environments and Hybrid Verifiers for Scaling Open-Weights SWE Agents","year":2025,"venue":"COLM 2025","authors":["Naman Jain","Jaskirat Singh","Manish Shetty","Liang Zheng","Koushik Sen","Ion Stoica"],"authors_zh":"Naman Jain, Jaskirat Singh, Manish Shetty, Liang Zheng, Koushik Sen, Ion Stoica","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["software-engineering","environments","unit-tests","rlvr","2025"],"status":"verified","priority":"必读","paper_type_zh":"程序化仓库环境、可执行验证器与软件 agent 训练数据","best_for_zh":"需要可重放软件工程环境、单测奖励或开放代码 agent 训练数据的研究者。","confidence":"high","one_line":["R2E-Gym procedurally turns repository commits into more than 8,000 executable SWE environments with test-based rewards for training and evaluating open agents.","R2E-Gym 将仓库提交程序化转换为 8,000 余个可执行 SWE 环境，并以单测奖励支持开放代码 agent 的训练与评测。"],"why":"It releases the runnable environment and terminal test reward rather than only issue–patch text pairs.","primary_link":"https://arxiv.org/abs/2504.07164","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/R2E-Gym/R2E-Gym"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/R2E-Gym/R2E-Gym-V1"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/R2E-Gym"},{"key":"project","label":["Project","项目主页"],"url":"https://r2e-gym.github.io/"}],"link_count":6,"sections":9},{"id":"r3-robust-rubric-agnostic-reward-models-2025","title":"R3: Robust Rubric-Agnostic Reward Models","year":2025,"venue":"NeurIPS 2025 LLM Evaluation Workshop","authors":["David Anugraha","Zilu Tang","Lester James V. Miranda","Hanyang Zhao","Mohammad Rifqi Farhansyah","Garry Kuwanto","Derry Wijaya","Genta Indra Winata"],"authors_zh":"David Anugraha、Zilu Tang、Lester James V. Miranda、Hanyang Zhao、Mohammad Rifqi Farhansyah 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","R3 让奖励模型根据输入量规生成可解释评分与理由，使奖励判断可控且可迁移。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2505.13388","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rubricreward/r3"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/rubricreward/R3-Dataset-20K"},{"key":"project","label":["Project","项目主页"],"url":"https://rubricreward.github.io/"}],"link_count":4,"sections":9},{"id":"rag-critic-automated-critic-guided-agentic-rag-2025","title":"RAG-Critic: Leveraging Automated Critic-Guided Agentic Workflow for Retrieval Augmented Generation","year":2025,"venue":"ACL 2025","authors":["Guanting Dong","Jiajie Jin","Xiaoxi Li","Yutao Zhu","Zhicheng Dou","Ji-Rong Wen"],"authors_zh":"Guanting Dong、Jiajie Jin、Xiaoxi Li、Yutao Zhu、Zhicheng Dou、Ji-Rong Wen","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","summarization"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"必读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["RAG-Error-Critic-100K is the paper's released feedback or evaluation resource.","100K RAG回答含正确性判断、细粒度错误层级与诊断文本，可复用于检索推理反馈。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2025.acl-long.179/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUC-NLPIR/RAG-Critic"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/dongguanting/RAG-Error-Critic-100K"}],"link_count":3,"sections":9},{"id":"rag-rewardbench-preference-alignment-2025","title":"RAG-RewardBench: Benchmarking Reward Models in Retrieval Augmented Generation for Preference Alignment","year":2025,"venue":"Findings of ACL 2025","authors":["Zhuoran Jin","Hongbang Yuan","Tianyi Men","Pengfei Cao","Yubo Chen","Jiexin Xu","Huaijun Li","Xiaojian Jiang","Kang Liu","Jun Zhao"],"authors_zh":"Zhuoran Jin, Hongbang Yuan, Tianyi Men, Pengfei Cao, Yubo Chen, Jiexin Xu, Huaijun Li, Xiaojian Jiang, Kang Liu, Jun Zhao","tracks":["preference_reward_feedback_data","audit_failure_contamination_verifier_attacks"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["retrieval-augmented-generation","preference-alignment","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"检索增强生成奖励模型评测与偏好数据基准论文","best_for_zh":"需要选择或审计检索增强生成系统的奖励模型，并覆盖引用、拒答和冲突证据情形的研究者。","confidence":"high","one_line":["RAG-RewardBench provides 1,485 preference pairs for testing reward models on retrieval-specific reasoning, citation, abstention, and conflict cases.","1,485组偏好对覆盖多跳推理、引用与冲突处理，以偏好标签训练或评估检索增强推理奖励。"],"why":"It makes RAG preference feedback inspectable at the document-and-response level rather than treating retrieval as ordinary chat.","primary_link":"https://aclanthology.org/2025.findings-acl.877/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/jinzhuoran/RAG-RewardBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jinzhuoran/RAG-RewardBench"}],"link_count":5,"sections":9},{"id":"ragferee-contextual-reward-models-2025","title":"RAGferee: Contextual Reward Models for RAG","year":2025,"venue":"EMNLP 2025","authors":["Andrei C. Coman","Ionut-Teodor Sorodoc","Leonardo F. R. Ribeiro","Bill Byrne","James Henderson","Adrià de Gispert"],"authors_zh":"Andrei C. Coman、Ionut-Teodor Sorodoc、Leonardo F. R. Ribeiro、Bill Byrne、James Henderson、Adrià de Gispert","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈数据审计","confidence":"","one_line":["RAGferee releases retrieval-grounded question answering contexts paired with competing answers and contextual quality choices for bounded preference learning, reward modeling, evaluation, and audit.","RAGferee 发布围绕其任务场景组织的候选回答比较与反馈记录，可用于有边界的偏好学习、奖励建模、评测和审计。"],"why":"The release supports feedback learning for 检索证据约束下的回答质量.","primary_link":"https://aclanthology.org/2025.emnlp-main.414/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/amazon-science/RAGferee"}],"link_count":2,"sections":9},{"id":"rationalyst-2025","title":"RATIONALYST: Pre-training Process-Supervision for Improving Reasoning","year":2025,"venue":"ACL","authors":["Dongwei Jiang","Guoxuan Wang","Yining Lu","Andrew Wang","Jingyu Zhang","Chuyu Liu","Benjamin Van Durme","Daniel Khashabi"],"authors_zh":"Dongwei Jiang、Guoxuan Wang、Yining Lu、Andrew Wang、Jingyu Zhang、Chuyu Liu、Benjamin Van Durme、Daniel Khashabi","tracks":["data_construction_open_release_recipes","process_trace_supervision_data"],"source_role":["construction_recipe","process_supervision","data_release"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft","process_supervision","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","search_substrate"],"domains":["mathematics","commonsense_reasoning","scientific_reasoning","logical_reasoning"],"tags":["implicit-rationales","process-supervision","self-supervised-data","the-pile","rationale-mining","future-token-prediction","step-selection"],"status":"partial","priority":"必读","paper_type_zh":"弱验证驱动的过程监督数据构建论文","best_for_zh":"研究隐式 rationale、弱验证器、过程监督、测试时步骤选择与数据发布审计的读者","confidence":"high","one_line":["Reports a 79K implicit-rationale mixture filtered by future-token predictiveness and trains an 8B process supervisor, while the linked 15178-row dataset leaves the full release unreconciled.","论文报告用后续 token 可预测性筛得约 7.9 万条隐式 rationale，并训练 8B 过程监督器；但官方链接数据集只有 15,178 行，完整发布尚未对齐。"],"why":"It makes a weak, scalable verifier contract concrete and shows why predictive filtering, source lineage, release completeness, and rationale validity must be audited separately.","primary_link":"https://aclanthology.org/2025.acl-long.1288/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JHU-CLSP/Rationalyst/tree/be0b1cf5c3fd1a642e054065fb02db382885153e"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Dongwei/reasoning_world_model"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Dongwei/Rationalyst_reasoning_datasets"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/JHU-CLSP/Rationalyst"}],"link_count":8,"sections":9},{"id":"realcritic-effectiveness-driven-critique-2025","title":"RealCritic: Towards Effectiveness-Driven Evaluation of Language Model Critiques","year":2025,"venue":"arXiv preprint","authors":["Zhengyang Tang","Ziniu Li","Zhenyang Xiao","Tian Ding","Ruoyu Sun","Benyou Wang","Dayiheng Liu","Fei Huang","Tianyu Liu","Bowen Yu","Junyang Lin"],"authors_zh":"Zhengyang Tang、Ziniu Li、Zhenyang Xiao、Tian Ding、Ruoyu Sun、Benyou Wang、Dayiheng Liu、Fei Huang、Tianyu Liu、Bowen Yu、Junyang Lin","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["RealCritic evaluates critiques in a closed loop through whether they lead to effective corrections.","以批评能否促成正确修正为闭环标准，评估自批评、交叉批评与迭代批评。"],"why":"RealCritic evaluates critiques in a closed loop through whether they lead to effective corrections.","primary_link":"https://arxiv.org/abs/2501.14492","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tangzhy/RealCritic"}],"link_count":2,"sections":9},{"id":"reasoning-gym-2025","title":"REASONING GYM: Reasoning Environments for Reinforcement Learning with Verifiable Rewards","year":2025,"venue":"NeurIPS 2025 Spotlight","authors":["Zafir Stojanovski","Oliver Stanley","Joe Sharratt","Richard Jones","Abdulhakeem Adefioye","Jean Kaddour","Andreas Köpf"],"authors_zh":"Zafir Stojanovski, Oliver Stanley, Joe Sharratt, Richard Jones, Abdulhakeem Adefioye, Jean Kaddour, Andreas Köpf","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["agent_environment","construction_recipe","verifier_reward","benchmark","infrastructure"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["algebra","algorithmic-reasoning","arithmetic","code","cognition","games","geometry","graph-reasoning","induction","logic"],"tags":["procedural-generation","reasoning-environment","rlvr","rule-based-verifier","dynamic-curriculum","synthetic-data","open-source","version-sensitive"],"status":"verified","priority":"必读","paper_type_zh":"程序化推理环境与 RLVR 构造基础设施","best_for_zh":"构建动态 RLVR 课程、可执行验证评测或版本化推理环境的研究者","confidence":"medium","one_line":["Reasoning Gym is an open library of more than 100 configurable procedural reasoning generators with task-specific executable rewards for dynamic RLVR training and evaluation.","Reasoning Gym 将 100 多种可配置程序化推理生成器与任务专用可执行奖励组合起来，使带版本的“生成器—verifier”而非固定语料库成为可复用 RLVR 数据对象。"],"why":"It makes the reusable reasoning-data object a versioned generator-verifier environment whose realized examples and rewards depend on code, configuration, seeds, scorer behavior, and curriculum state.","primary_link":"https://openreview.net/forum?id=GqYSunGmp7","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/open-thought/reasoning-gym"}],"link_count":4,"sections":9},{"id":"reasonmed-2025","title":"ReasonMed: A 370K Multi-Agent Generated Dataset for Advancing Medical Reasoning","year":2025,"venue":"EMNLP","authors":["Yu Sun","Xingyu Qian","Weiwen Xu","Hao Zhang","Chenghao Xiao","Long Li","Deli Zhao","Wenbing Huang","Tingyang Xu","Qifeng Bai","Yu Rong"],"authors_zh":"Yu Sun、Xingyu Qian、Weiwen Xu、Hao Zhang、Chenghao Xiao、Long Li、Deli Zhao、Wenbing Huang、Tingyang Xu、Qifeng Bai、Yu Rong","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline"],"domains":["medical_reasoning","medical_question_answering"],"tags":["medical-reasoning","multi-agent-data-generation","chain-of-thought","llm-as-judge","difficulty-routing","expert-audit","open-data"],"status":"verified","priority":"必读","paper_type_zh":"医学推理数据发布与多教师难度路由构造配方","best_for_zh":"关注医学推理 SFT、多教师生成、LLM 判定器、难度路由、专家审计与高风险数据治理的研究者","confidence":"high","one_line":["Routes nine candidate paths per medical MCQ through answer-key-conditioned verification and easy/medium/difficult repair, then releases three unequal SFT views totaling 1,111,555 rows.","每道医学选择题拟生成九条路径，经答案键条件下的验证与难度路由修复后，发布三套不等长的 SFT 视图，共 1,111,555 行。"],"why":"Provides a concrete high-risk-domain construction recipe whose open rows also expose judge dependence, missing lineage, count reconciliation, and medical-validity limits.","primary_link":"https://aclanthology.org/2025.emnlp-main.1344/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/alibaba-damo-academy/ReasonMed"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lingshu-medical-mllm/ReasonMed"}],"link_count":6,"sections":9},{"id":"refactorbench-stateful-agent-reasoning-2025","title":"RefactorBench: Evaluating Stateful Reasoning in Language Agents Through Code","year":2025,"venue":"ICLR 2025","authors":["Dhruv Gautam","Spandan Garg","Jinu Jang","Neel Sundaresan","Roshanak Zilouchian Moghaddam"],"authors_zh":"Dhruv Gautam、Spandan Garg、Jinu Jang、Neel Sundaresan、Roshanak Zilouchian Moghaddam","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-agents"],"tags":["code-agents","refactoring","repository-tests","2025"],"status":"verified","priority":"可读","paper_type_zh":"代码重构代理基准与可执行评测集","best_for_zh":"研究代码代理、跨文件状态推理和测试驱动评测的研究者。","confidence":"high","one_line":["RefactorBench tests language agents' persistent reasoning over cross-file state and code transformations through executable refactoring tasks.","RefactorBench 以可执行重构任务测试语言代理对跨文件状态与代码变换的持续推理。"],"why":"RefactorBench tests language agents' persistent reasoning over cross-file state and code transformations through executable refactoring tasks.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/6b44ee74539ea77d6a0d50d468724371-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/RefactorBench"}],"link_count":2,"sections":9},{"id":"reference-guided-verdict-2025","title":"Reference-Guided Verdict: LLMs-as-Judges in Automatic Evaluation of Natural Language Generation","year":2025,"venue":"WiNLP 2025","authors":["Sher Badshah","Hassan Sajjad"],"authors_zh":"Sher Badshah, Hassan Sajjad","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"可读","paper_type_zh":"有参考答案的多 LLM judge 自由问答评测研究","best_for_zh":"需要为开放式问答建立可校准、多裁判自动评测的研究者。","confidence":"medium","one_line":["Studies reference-guided multi-judge evaluation and reliability tradeoffs.","以参考答案引导多个 LLM judge 多数投票，评估自由问答回答与人工判决的一致性。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.winlp-main.37/","links":[],"link_count":1,"sections":9},{"id":"regenesis-2025","title":"ReGenesis: LLMs can Grow into Reasoning Generalists via Self-Improvement","year":2025,"venue":"ICLR 2025 Oral","authors":["Xiangyu Peng","Congying Xia","Xinyi Yang","Caiming Xiong","Chien-Sheng Wu","Chen Xing"],"authors_zh":"Xiangyu Peng, Congying Xia, Xinyi Yang, Caiming Xiong, Chien-Sheng Wu, Chen Xing","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["math","logic","commonsense","natural-language-inference"],"tags":["self-improvement","self-generated-reasoning","abstract-to-concrete","reasoning-guidelines","reasoning-structures","answer-filtering","ground-truth-hints","supervised-fine-tuning"],"status":"partial","priority":"可读","paper_type_zh":"自生成推理轨迹构造与 SFT 配方","best_for_zh":"研究推理数据合成、答案筛选、自训练与跨任务泛化的读者","confidence":"high","one_line":["ReGenesis transforms 25 general reasoning guidelines into task-specific structures and answer-selected paths, retaining at most five trajectories per instruction for SFT.","ReGenesis 用 25 条通用推理准则逐层生成任务准则、结构与答案筛选轨迹，展示了跨任务自生成 SFT 配方，但终局匹配不能验证步骤且官方语料与代码尚未发布。"],"why":"For the Data Construction track, it makes pre-trace planning objects and retry lineage explicit, while also showing why terminal-answer success, hidden upstream hints, and unreleased corpora require separate quality audits.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/9c77f2ce42151b2c2e26d2cf47f99564-Abstract-Conference.html","links":[],"link_count":5,"sections":9},{"id":"group-preference-reward-shaping-2025","title":"Reinforcement Learning for Large Language Models via Group Preference Reward Shaping","year":2025,"venue":"EMNLP 2025","authors":["Huaisheng Zhu","Siyuan Xu","Hangfan Zhang","Teng Xiao","Zhimeng Guo","Shijie Zhou","Shuyue Hu","Vasant G. Honavar"],"authors_zh":"Huaisheng Zhu, Siyuan Xu, Hangfan Zhang, Teng Xiao, Zhimeng Guo, Shijie Zhou, Shuyue Hu, Vasant G. Honavar","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"可读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["GPRS replaces raw reward optimization with within-group preference shaping to reduce reward-model sensitivity.","以组内偏好塑形替代原始分数优化，降低无 critic RLHF 对奖励模型质量的敏感性。"],"why":"It exposes how the form of a reward signal can affect post-training stability and downstream alignment claims.","primary_link":"https://aclanthology.org/2025.emnlp-main.1085/","links":[],"link_count":1,"sections":9},{"id":"reinforcement-learning-teachers-test-time-scaling-2025","title":"Reinforcement Learning Teachers of Test Time Scaling","year":2025,"venue":"NeurIPS","authors":["Edoardo Cetin","Tianyu Zhao","Yujin Tang"],"authors_zh":"Edoardo Cetin, Tianyu Zhao, Yujin Tang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["sft","distillation","reward_modeling","rlvr"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","coding","science","general-reasoning"],"tags":["reasoning-distillation","teacher-traces","reinforcement-learned-teacher","student-conditioned-reward","grpo","cold-start-data","open-code"],"status":"partial","priority":"可读","paper_type_zh":"推理数据构建方案与强化学习教师研究","best_for_zh":"设计教师轨迹、学生条件奖励、推理蒸馏或强化学习冷启动流程的研究者","confidence":"high","one_line":["RLT trains a Qwen2.5-7B-Instruct teacher with dense frozen-student likelihood and explanation-KL feedback to generate raw explanations from known question-solution pairs for downstream distillation.","RLT 以冻结学生的答案似然和解释 KL 反馈训练 Qwen2.5-7B-Instruct 教师，让其基于已知问题—答案对生成供蒸馏的原始解释，但未发布轨迹数据或明确的教师权重。"],"why":"It makes the teacher-conditioning and feedback contract unusually concrete, while showing that student-specific rewards, answer-conditioned explanations, missing trace releases, and missing teacher weights remain central audit boundaries.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/9a6b278218966499194491f55ccf8b75-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SakanaAI/RLT"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/SakanaAI/reinforcement-learning-teachers"},{"key":"project","label":["Project","项目主页"],"url":"https://sakana.ai/rlt/"}],"link_count":7,"sections":9},{"id":"repanda-pandas-tabular-verification-reasoning-2025","title":"RePanda: Pandas-powered Tabular Verification and Reasoning","year":2025,"venue":"ACL 2025","authors":["Atoosa Malemir Chegini","Keivan Rezaei","Hamid Eghbalzadeh","Soheil Feizi"],"authors_zh":"Atoosa Malemir Chegini, Keivan Rezaei, Hamid Eghbalzadeh, Soheil Feizi","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["tabular-reasoning","fact-verification"],"tags":["tables","pandas","fact-verification","execution","acl-2025"],"status":"verified","priority":"可读","paper_type_zh":"表格事实核验与可执行查询数据集","best_for_zh":"需要表格推理程序监督、可解释事实核验或 pandas 执行验证的研究者。","confidence":"high","one_line":["RePanda converts table claims into executable pandas queries so fact verification is determined by rerunnable table operations rather than a black-box label.","RePanda 将表格断言编译为可执行的 pandas 查询，使事实核验可由重运行的表格操作决定。"],"why":"The record exposes the executable operation that establishes the table-grounded answer.","primary_link":"https://aclanthology.org/2025.acl-long.1549/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AtoosaChegini/PanTabFact"}],"link_count":4,"sections":9},{"id":"replay-failures-as-successes-2025","title":"Replay Failures as Successes: Sample-Efficient Reinforcement Learning for Instruction Following","year":2025,"venue":"arXiv preprint","authors":["Kongcheng Zhang","Qi Yao","Shunyu Liu","Wenjian Zhang","Min Cen","Yang Zhou","Wenkai Fang","Yiru Zhao","Baisheng Lai","Mingli Song"],"authors_zh":"Kongcheng Zhang、Qi Yao、Shunyu Liu、Wenjian Zhang、Min Cen、Yang Zhou、Wenkai Fang、Yiru Zhao、Baisheng Lai、Mingli Song","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","pairwise_preference","scalar_reward"],"training_use":["preference_learning","rlvr"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["reasoning"],"tags":["failure-replay","hindsight-replay","instruction-following","constraints","rl"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["HiR selects partially successful failed rollouts, rewrites them from satisfied constraints, and replays them during RL.","HiR 挑出部分成功的失败采样，按其已满足的约束改写它们，并在强化学习中回放。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"Selection and rewriting signals need retention to audit failure-to-data transformations.","primary_link":"https://arxiv.org/abs/2512.23457","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sastpg/HIR"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/sastpg/HIR-16K"}],"link_count":4,"sections":9},{"id":"researchrubrics-deep-research-agents-2025","title":"ResearchRubrics: A Benchmark of Prompts and Rubrics For Evaluating Deep Research Agents","year":2025,"venue":"ICLR 2026","authors":["Manasi Sharma","Chen Bo Calvin Zhang","Chaithanya Bandi","Clinton J. Wang","Ankit Aich","Huy Nghiem","Tahseen Rabbani","Ye Htet","Brian Jang","Sumana Basu","Aishwarya Balwani","Denis Peskoff","Marcos Ayestaran","Sean M. Hendryx","Brad Kenstler","Bing Liu"],"authors_zh":"Manasi Sharma, Chen Bo Calvin Zhang, Chaithanya Bandi, Clinton J. Wang, Ankit Aich, Huy Nghiem, Tahseen Rabbani, Ye Htet, Brian Jang, Sumana Basu, Aishwarya Balwani, Denis Peskoff, Marcos Ayestaran, Sean M. Hendryx, Brad Kenstler, Bing Liu","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["deep_research","factuality"],"tags":["deep_research","rubric","llm_judge","benchmark"],"status":"verified","priority":"必读","paper_type_zh":"深度研究 Agent 的人类 rubric 评测基准","best_for_zh":"需要评估多文档研究、长报告 Judge 或证据支撑推理的研究者。","confidence":"high","one_line":["ResearchRubrics pairs realistic research prompts with human-written criteria for granular agent evaluation.","ResearchRubrics 用人类撰写并复核的细粒度准则，衡量深度研究 Agent 的事实依据、推理与表达。"],"why":"It evaluates justification and synthesis rather than only short factual retrieval.","primary_link":"https://arxiv.org/abs/2511.07685","links":[{"key":"code","label":["Code","代码"],"url":"https://scale.com/research/researchrubrics"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScaleAI/researchrubrics"}],"link_count":4,"sections":9},{"id":"rethinking-improving-autoformalization-2025","title":"Rethinking and Improving Autoformalization: Towards a Faithful Metric and a Dependency Retrieval-Based Approach","year":2025,"venue":"ICLR 2025","authors":["Qi Liu","Xinhao Zheng","Xudong Lu","Qinxiang Cao","Junchi Yan"],"authors_zh":"Qi Liu, Xinhao Zheng, Xudong Lu, Qinxiang Cao, Junchi Yan","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["formal-mathematics","lean4","autoformalization"],"tags":["autoformalization","lean4","beq","retrieval","iclr-2025"],"status":"verified","priority":"可读","paper_type_zh":"语义验证与依赖检索的 Lean 自动形式化基准","best_for_zh":"研究 Lean 自动形式化、语义等价验证或定理检索的研究者。","confidence":"high","one_line":["This work couples a bidirectional Lean equivalence verifier with dependency retrieval and the 961-item Con-NF benchmark to measure and improve faithful autoformalization.","论文以双向 BEq 验证、依赖检索和 961 题 Con-NF，将自动形式化的评测从可编译性推进到可验证的语义等价。"],"why":"It distinguishes semantic correctness from compilability and makes library dependencies explicit supervision for formalization.","primary_link":"https://openreview.net/forum?id=hUb2At2DsQ","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Purewhite2019/rethinking_autoformalization"}],"link_count":3,"sections":9},{"id":"rethink-selection-at-scale-2025","title":"Rethinking Data Selection at Scale: Random Selection is Almost All You Need","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Tingyu Xia","Bowen Yu","Kai Dang","An Yang","Yuan Wu","Yuan Tian","Yi Chang","Junyang Lin"],"authors_zh":"Tingyu Xia、Bowen Yu、Kai Dang、An Yang、Yuan Wu、Yuan Tian、Yi Chang、Junyang Lin","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","optimizer_scaffold","scaling_report","release_audit"],"domains":["general_reasoning","mathematical_reasoning","code_reasoning","instruction_following","reasoning_data_selection"],"tags":["sft-data-selection","random-selection","self-scoring","data-diversity","token-length","k-means","scaling-study","negative-result","partial-code-release"],"status":"partial","priority":"必读","paper_type_zh":"大规模 SFT data selection 的 construction recipe、scaling negative result 与 release audit","best_for_zh":"研究 SFT subset selection、self-scoring、diversity、token-length heuristic、随机基线及 data-selection 可复现性的读者","confidence":"medium","one_line":["The study selects 10K or 50K OpenHermes and English WildChat conversations with gradient, loss, uncertainty, diversity, compression, random, or token-length rules, then compares Qwen2-7B and Llama3-8B SFT across five benchmark families.","该 Findings of EMNLP 2025 研究在 OpenHermes 与 English WildChat 两个大池上比较六种 self-scoring selector、五次 random control 及 10K/50K SFT subset，显示复杂选择很少稳定胜过随机；其 token-length+K-means 实用 recipe 虽强，但公开代码只处理前 100 个 embedding 并选择 cluster center，不能复现论文的按 cluster 比例选择最长样本。"],"why":"It tests whether sophisticated self-scoring survives million-scale pools and real compute limits, while the missing subset IDs, code/recipe mismatch, contamination audit, license reconciliation, and run manifest show what a credible negative scaling claim must release.","primary_link":"https://aclanthology.org/2025.findings-emnlp.146/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xiatingyu/SFT-DataSelection-at-scale"},{"key":"data","label":["Data","数据"],"url":"https://www.modelscope.cn/datasets/xiatingyu/wildchat-en"}],"link_count":9,"sections":9},{"id":"dco-test-time-compute-2025","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Feng Chen","Allan Raventós","Nan Cheng","Surya Ganguli","Shaul Druckmann"],"authors_zh":"Feng Chen、Allan Raventós、Nan Cheng、Surya Ganguli、Shaul Druckmann（机构：Stanford University、University of Michigan）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["test_time_compute","sft","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","formal-reasoning"],"tags":["test-time-compute","pass-at-n","confidence","supervised-finetuning","theorem-proving","mathematical-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时计算对齐的微调研究（NeurIPS 2025）","best_for_zh":"研究推理模型训练目标如何与 pass@N 或证明搜索预算共同设计的读者。","confidence":"high","one_line":["DCO aligns fine-tuning with a chosen pass@N test-time budget by limiting overconfidence and preserving useful answer diversity.","DCO 通过限制过度置信，使微调目标与预定的 pass@N 测试时采样预算对齐，并保留有用的答案多样性。"],"why":"It demonstrates that a training loss optimized for one sample can reduce performance when the deployed system relies on many samples.","primary_link":"https://arxiv.org/abs/2502.07154","links":[],"link_count":3,"sections":9},{"id":"rethinking-optimal-verification-granularity-2025","title":"Rethinking Optimal Verification Granularity for Compute-Efficient Test-Time Scaling","year":2025,"venue":"NeurIPS 2025","authors":["Hao Mark Chen","Guanxi Lu","Yasuyuki Okoshi","Zhiwen Mo","Masato Motomura","Hongxiang Fan"],"authors_zh":"Hao Mark Chen、Guanxi Lu、Yasuyuki Okoshi、Zhiwen Mo、Masato Motomura、Hongxiang Fan","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer"],"domains":["mathematics","reasoning"],"tags":["vg-search","verifier","prm","beam-search","compute-efficiency"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["VG-Search controls verifier cadence to trade prefix pruning against inference cost.","VG-Search 通过控制验证器的调用频率，在前缀剪枝收益与推理开销之间做权衡。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"Verification frequency itself is a trace/compute allocation policy.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/8011b23e1dc3f57e1b6211ccad498919-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hmarkc/VG-Search"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/blog/search-and-learn"}],"link_count":5,"sections":9},{"id":"retrieval-augmented-process-reward-model-generalizable-math-2025","title":"Retrieval-Augmented Process Reward Model for Generalizable Mathematical Reasoning","year":2025,"venue":"Findings of the Association for Computational Linguistics: ACL 2025","authors":["Jiachen Zhu","Congmin Zheng","Jianghao Lin","Kounianhua Du","Ying Wen","Yong Yu","Jun Wang","Weinan Zhang"],"authors_zh":"Jiachen Zhu、Congmin Zheng、Jianghao Lin、Kounianhua Du、Ying Wen 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["math_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"检索增强过程奖励建模与数学步骤监督论文","best_for_zh":"研究过程奖励模型的跨题型泛化与检索辅助步骤判定。","confidence":"high","one_line":["RetrievalPRM releases retrieval-augmented step supervision so process reward models can judge unfamiliar mathematical reasoning patterns more reliably.","RetrievalPRM 发布检索增强的步骤监督数据，使过程奖励模型能更可靠地判断陌生的数学推理模式。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://aclanthology.org/2025.findings-acl.444/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/gebro13/RetrievalPRM_Dataset"}],"link_count":3,"sections":9},{"id":"revise-intrinsic-self-verification-2025","title":"ReVISE: Learning to Refine at Test-Time via Intrinsic Self-Verification","year":2025,"venue":"ICML 2025","authors":["Hyunseok Lee","Seunghyuk Oh","Jaehyung Kim","Jinwoo Shin","Jihoon Tack"],"authors_zh":"Hyunseok Lee、Seunghyuk Oh、Jaehyung Kim、Jinwoo Shin、Jihoon Tack","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["pairwise_preference","full_episode"],"training_use":["preference_learning","test_time_compute"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["reasoning"],"tags":["self-verification","self-correction","preference-learning","failed-traces"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["ReVISE converts failed and successful reasoning paths into preference pairs for intrinsic verification and refinement.","ReVISE 把失败与成功的推理路径转成偏好对，用于内生的自我验证与改写。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"It retains both successful and failed trajectories rather than only final answers.","primary_link":"https://proceedings.mlr.press/v267/lee25ab.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/seunghyukoh/revise"}],"link_count":3,"sections":9},{"id":"guru-reasoning360-2025","title":"Revisiting Reinforcement Learning for LLM Reasoning from A Cross-Domain Perspective","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Jorge (Zhoujun) Cheng","Shibo Hao","Tianyang Liu","Fan Zhou","Yutao Xie","Feng Yao","Yuexin Bian","Nilabjo Dey","Yonghao Zhuang","Yuheng Zha","Yi Gu","Kun Zhou","Yuqi Wang","Yuan Li","Richard Fan","Jianshu She","Chengqian Gao","Abulhair Saparov","Taylor W. Killian","Haonan Li","Mikhail Yurochkin","Eric P. Xing","Zhengzhong Liu","Zhiting Hu"],"authors_zh":"Jorge (Zhoujun) Cheng, Shibo Hao, Tianyang Liu, Fan Zhou, Yutao Xie, Feng Yao, Yuexin Bian, Nilabjo Dey, Yonghao Zhuang, Yuheng Zha, Yi Gu, Kun Zhou, Yuqi Wang, Yuan Li, Richard Fan, Jianshu She, Chengqian Gao, Abulhair Saparov, Taylor W. Killian, Haonan Li, Mikhail Yurochkin, Eric P. Xing, Zhengzhong Liu, Zhiting Hu","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning","code","scientific_reasoning","logical_reasoning","simulation","tabular_reasoning"],"tags":["guru","reasoning360","multi-domain","rlvr","difficulty-filtering","deduplication","rule-based-verifier","code-execution","model-verifier","open-data"],"status":"partial","priority":"必读","paper_type_zh":"多领域 RLVR 数据发布与构建研究","best_for_zh":"适合构建、复现或审计跨领域强化学习推理数据与验证器契约的研究者。","confidence":"high","one_line":["GURU filters about 684.9K public and synthetic candidates into 91,909 training rows spanning mathematics, code, science, logic, simulation, and tables, with rule, execution, or model-based rewards.","GURU 将约 68.49 万条候选过滤为 91,909 条六领域 RLVR 训练记录，并为不同领域配置规则、执行或模型验证奖励。"],"why":"It exposes a comparatively complete cross-domain RL data contract while leaving material uncertainty around semantic contamination, verifier error, mixed-license lineage, rejected candidates, and absent policy rollouts.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/c2f71567cd53464161cab3336e8fc865-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LLM360/Reasoning360"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LLM360/guru-RL-92k"},{"key":"project","label":["Project","项目主页"],"url":"https://guru-reasoning.github.io/"}],"link_count":5,"sections":9},{"id":"rewardbench-2-advancing-reward-model-evaluation-2025","title":"RewardBench 2: Advancing Reward Model Evaluation","year":2025,"venue":"arXiv 2025","authors":["Saumya Malik","Valentina Pyatkin","Sander Land","Jacob Morrison","Noah A. Smith","Hannaneh Hajishirzi","Nathan Lambert"],"authors_zh":"Saumya Malik, Valentina Pyatkin, Sander Land, Jacob Morrison, Noah A. Smith, Hannaneh Hajishirzi, Nathan Lambert","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","reasoning-evaluation","preference-learning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"多技能奖励模型评测与偏好数据基准论文","best_for_zh":"需要在不复用下游测试提示的前提下评估奖励模型并分析其与推理或 RLHF 后效关系的研究者。","confidence":"high","one_line":["RewardBench 2 adds 1,865 new human-prompt preference examples for harder, multi-skill reward-model evaluation linked to downstream use.","1,865条新偏好评测覆盖数学等任务，以相对偏好检验 RM，适合验证通用推理奖励的泛化。"],"why":"It limits prompt reuse from downstream benchmarks while testing whether reward ranking predicts useful inference-time and RLHF behavior.","primary_link":"https://arxiv.org/abs/2506.01937","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/reward-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/reward-bench-2"}],"link_count":4,"sections":9},{"id":"rewardbench-evaluating-reward-models-2025","title":"RewardBench: Evaluating Reward Models for Language Modeling","year":2025,"venue":"Findings of NAACL 2025","authors":["Nathan Lambert","Valentina Pyatkin","Jacob Morrison","LJ Miranda","Bill Yuchen Lin","Khyathi Chandu","Nouha Dziri","Sachin Kumar","Tom Zick","Yejin Choi","Noah A. Smith","Hannaneh Hajishirzi"],"authors_zh":"Nathan Lambert, Valentina Pyatkin, Jacob Morrison, LJ Miranda, Bill Yuchen Lin, Khyathi Chandu, Nouha Dziri, Sachin Kumar, Tom Zick, Yejin Choi, Noah A. Smith, Hannaneh Hajishirzi","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","reasoning-evaluation","safety"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"通用奖励模型评测基准与偏好数据论文","best_for_zh":"需要用覆盖对话、推理和安全的成对偏好记录对比奖励模型，或复现 RewardBench 评测协议的研究者。","confidence":"high","one_line":["RewardBench supplies 2,985 prompt-chosen-rejected comparisons for evaluating reward models on chat, reasoning, safety, and subtle correctness failures.","2,985组偏好对含推理与代码子集，提供 chosen/rejected 监督，是通用奖励模型的基础对照。"],"why":"It is a reusable baseline for testing whether a reward model recognizes the concrete reason one response should be preferred.","primary_link":"https://aclanthology.org/2025.findings-naacl.96/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/reward-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/reward-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/allenai/reward-bench"}],"link_count":6,"sections":9},{"id":"rewarding-graph-reasoning-process-generalized-reasoners-2025","title":"Rewarding Graph Reasoning Process makes LLMs more Generalized Reasoners","year":2025,"venue":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD 2025)","authors":["Miao Peng","Nuo Chen","Zongrui Suo","Jia Li"],"authors_zh":"Miao Peng、Nuo Chen、Zongrui Suo、Jia Li","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["graph_reasoning","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"图推理过程奖励模型、自动步骤标注与监督数据论文","best_for_zh":"训练或评测图算法推理的步骤判别、过程奖励和跨领域泛化。","confidence":"high","one_line":["GraphSilo provides 118,189 graph-reasoning solution pairs with automatically derived step labels for training GraphPRM across 13 tasks.","GraphSilo 提供 11.8 万余个图推理解答对及自动生成的步骤标签，用于在 13 类任务上训练 GraphPRM。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://doi.org/10.1145/3711896.3737109","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GraphPRM/GraphSilo"}],"link_count":3,"sections":9},{"id":"rewordbench-2025","title":"reWordBench: Benchmarking and Improving the Robustness of Reward Models with Transformed Inputs","year":2025,"venue":"EMNLP 2025","authors":["Zhaofeng Wu","Michihiro Yasunaga","Andrew Cohen","Yoon Kim","Asli Celikyilmaz","Marjan Ghazvininejad"],"authors_zh":"Zhaofeng Wu, Michihiro Yasunaga, Andrew Cohen, Yoon Kim, Asli Celikyilmaz, Marjan Ghazvininejad","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"污染、验证器、奖励或评测可靠性审计","best_for_zh":"需要核查推理数据与自动评测可靠性的研究者。","confidence":"medium","one_line":["Meaning-preserving input transformations audit whether reward models rely on brittle shortcuts.","reWordBench用语义或排序不变改写揭示奖励模型捷径，并以同分改写训练提升鲁棒性。"],"why":"It adds a concrete reliability or failure-mode evaluation surface to Track 13.","primary_link":"https://aclanthology.org/2025.emnlp-main.167/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/reward-bench"}],"link_count":2,"sections":9},{"id":"rlaif-v-open-source-ai-feedback-2025","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","year":2025,"venue":"CVPR 2025","authors":["Tianyu Yu","Haoye Zhang"],"authors_zh":"Tianyu Yu、Haoye Zhang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness releases open-source visual-language-model feedback over image-text responses; synthetic labels target hallucination and trustworthiness for preference learning, reward modeling, evaluation, and feedback audit.","开放视觉语言模型生成的图文偏好反馈，用于降低视觉幻觉并训练多模态奖励模型。"],"why":"开放视觉语言模型生成的图文偏好反馈，用于降低视觉幻觉并训练多模态奖励模型。","primary_link":"https://arxiv.org/abs/2405.17220","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RLHF-V/RLAIF-V"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openbmb/RLAIF-V-Dataset"}],"link_count":3,"sections":9},{"id":"rlvr-world-2025","title":"RLVR-World: Training World Models with Reinforcement Learning","year":2025,"venue":"NeurIPS 2025","authors":["Jialong Wu","Shaofeng Yin","Ningya Feng","Mingsheng Long"],"authors_zh":"Jialong Wu、Shaofeng Yin、Ningya Feng、Mingsheng Long","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","data_release","verifier_reward"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["world_modeling","text_games","web_navigation","robotics"],"tags":["rlvr","world-model","verifiable-reward","grpo","state-transition","webarena","robot-trajectories","data-disclosure-ledger","neurips-2025"],"status":"partial","priority":"必读","paper_type_zh":"多模态世界模型的 RLVR 训练配方、反馈契约与数据披露审计","best_for_zh":"关注可验证奖励、世界模型、智能体环境轨迹、数据血缘和后训练审计的研究者与工程人员","confidence":"high","one_line":["Releases code, two derived language-transition datasets and RLVR checkpoints for world models whose programmatic rewards are exact state/item agreement or frame-level L1 plus LPIPS, while upstream lineage and audit artifacts remain partial.","RLVR-World 发布了代码、两个派生语言状态转移数据集和 RLVR 检查点；其文本、网页与视频世界模型分别采用精确状态/条目一致性或 L1 加 LPIPS 反馈，但上游数据血缘、权利、切分和审计工件仍不完整。"],"why":"It exposes feedback contracts and operational details that are usually obscured in RLVR reports, and it shows why a released model/data collection still needs separate provenance, split and verifier-validity auditing before reuse.","primary_link":"https://arxiv.org/abs/2505.13934","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thuml/RLVR-World"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/thuml/rlvr-world"},{"key":"project","label":["Project","项目主页"],"url":"https://thuml.github.io/RLVR-World/"}],"link_count":6,"sections":9},{"id":"rm-bench-subtlety-style-2025","title":"RM-Bench: Benchmarking Reward Models of Language Models with Subtlety and Style","year":2025,"venue":"ICLR 2025","authors":["Yantao Liu","Zijun Yao","Rui Min","Yixin Cao","Lei Hou","Juanzi Li"],"authors_zh":"Yantao Liu, Zijun Yao, Rui Min, Yixin Cao, Lei Hou, Juanzi Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","reasoning-evaluation","style-bias"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"奖励模型细微内容差异与风格偏差评测基准论文","best_for_zh":"需要识别奖励模型是否受写作风格干扰、能否分辨细微但关键内容差异的研究者。","confidence":"high","one_line":["RM-Bench tests reward models on 11,943 controlled comparisons that separate subtle content quality from style bias.","1,327提示、11,943偏好比较含数学和代码，以细微内容差异检验奖励是否真正识别推理质量。"],"why":"It exposes whether a reward model is responding to reasoning-relevant content rather than superficial writing style.","primary_link":"https://openreview.net/forum?id=QEHrmQPBdd","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/RM-Bench"}],"link_count":4,"sections":9},{"id":"rmb-comprehensive-reward-model-benchmark-2025","title":"RMB: Comprehensively Benchmarking Reward Models in LLM Alignment","year":2025,"venue":"ICLR 2025","authors":["Enyu Zhou","Guodong Zheng","Binghai Wang","Zhiheng Xi","Shihan Dou","Rong Bao","Wei Shen","Limao Xiong","Jessica Fan","Yurong Mou","Rui Zheng","Tao Gui","Qi Zhang","Xuanjing Huang"],"authors_zh":"Enyu Zhou, Guodong Zheng, Binghai Wang, Zhiheng Xi, Shihan Dou, Rong Bao, Wei Shen, Limao Xiong, Jessica Fan, Yurong Mou, Rui Zheng, Tao Gui, Qi Zhang, Xuanjing Huang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reward-modeling","preference-learning","evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"多场景奖励模型评测、成对偏好与 Best-of-N 基准论文","best_for_zh":"需要在现实场景中同时评估奖励模型的成对排序和 Best-of-N 选择能力的研究者。","confidence":"high","one_line":["RMB evaluates reward models across 49 real-world scenarios with both pairwise and Best-of-N preference tests.","49类真实场景、约1.8万偏好比较，同时测 pairwise 与 Best-of-N，含代码和推理场景的奖励反馈。"],"why":"It connects offline reward-model comparison to the two selection modes commonly used in alignment workflows.","primary_link":"https://openreview.net/forum?id=kmgrlG9TR0","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Zhou-Zoey/RMB-Reward-Model-Benchmark"}],"link_count":4,"sections":9},{"id":"robomonkey-vla-2025","title":"RoboMonkey: Scaling Test-Time Sampling and Verification for Vision-Language-Action Models","year":2025,"venue":"CoRL 2025","authors":["Jacky Kwok","Christopher Agia","Rohan Sinha","Matt Foutter","Shulu Li","Ion Stoica","Azalia Mirhoseini","Marco Pavone"],"authors_zh":"Jacky Kwok、Christopher Agia、Rohan Sinha、Matt Foutter、Shulu Li、Ion Stoica、Azalia Mirhoseini、Marco Pavone","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","scalar_reward"],"training_use":["reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","scaling_report"],"domains":["embodied_ai","visual_reasoning","household_robotics"],"tags":["robomonkey","vla","action-verification","robotics"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["RoboMonkey ranks sampled VLA actions with a verifier trained on RMSE-derived preferences.","RoboMonkey 用基于均方根误差偏好训练的验证器，对采样出的视觉语言动作候选排序。（论文未披露的发布、回放与审计细节保留为未知。）"],"why":"It separates action proposal, proxy preference, ranking, and execution trace contracts.","primary_link":"https://proceedings.mlr.press/v305/kwok25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/robomonkey-vla/RoboMonkey"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/robomonkey-vla/action_preference_bridge"},{"key":"project","label":["Project","项目主页"],"url":"https://robomonkey-vla.github.io/"}],"link_count":6,"sections":9},{"id":"gammapo-dynamic-margin-2025","title":"Robust Preference Optimization via Dynamic Target Margins","year":2025,"venue":"Findings of ACL 2025","authors":["Jie Sun","Junkang Wu","Jiancan Wu","Zhibo Zhu","Xingyu Lu","Jun Zhou","Lintao Ma","Xiang Wang"],"authors_zh":"Jie Sun 等（蚂蚁集团、上海数据科学重点实验室、国立新加坡大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_learning"],"tags":["preference-optimization","target-margin","noisy-preferences","dynamic-weighting"],"status":"verified","priority":"可读","paper_type_zh":"动态目标边界的偏好优化研究","best_for_zh":"适合已拥有偏好回答对、但需要降低模糊标签影响且不希望直接删样本的读者。","confidence":"high","one_line":["γ-PO converts each preference pair's reward margin into an instance-specific target, strengthening clear supervision and smoothing ambiguous supervision.","γ-PO 将每个偏好对的奖励间隔转为动态目标边界，强化清晰监督并平滑模糊监督。"],"why":"It treats confidence in a preference record as an optimization-time data-use decision instead of a fixed property of the corpus.","primary_link":"https://aclanthology.org/2025.findings-acl.282/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sunjie279/gammaPO"}],"link_count":3,"sections":9},{"id":"rollout-roulette-particle-scaling-2025","title":"Rollout Roulette: A Probabilistic Inference Approach to Inference-Time Scaling of LLMs using Particle-Based Monte Carlo Methods","year":2025,"venue":"NeurIPS 2025","authors":["Isha Puri","Shivchander Sudalairaj","Guangxuan Xu","Kai Xu","Akash Srivastava"],"authors_zh":"Isha Puri、Shivchander Sudalairaj、Guangxuan Xu、Kai Xu、Akash Srivastava（机构：麻省理工学院计算机科学与人工智能实验室、Red Hat AI Innovation）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-compute","particle-filtering","prm","search","rollout-allocation"],"status":"verified","priority":"必读","paper_type_zh":"概率推理测试时扩展与搜索研究","best_for_zh":"在不完美过程奖励模型下设计稳健多轨迹推理的读者。","confidence":"high","one_line":["Rollout Roulette uses particle filtering to preserve diverse reasoning trajectories under process-reward uncertainty.","Rollout Roulette 利用粒子滤波在过程奖励不确定时保留多样化的推理轨迹。"],"why":"It turns reward-guided reasoning search into uncertainty-aware trajectory allocation rather than greedy pruning.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/e55c675d3230dbc3bf24c986d6685632-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Red-Hat-AI-Innovation-Team/its_hub"},{"key":"project","label":["Project","项目主页"],"url":"https://probabilistic-inference-scaling.github.io/"}],"link_count":5,"sections":9},{"id":"rottenreviews-peer-review-quality-2025","title":"RottenReviews: Benchmarking Review Quality with Human and LLM-Based Judgments","year":2025,"venue":"CIKM 2025","authors":["Sajad Ebrahimi","Soroush Sadeghian","Ali Ghorbanpour","Negar Arabzadeh","Sara Salamat","Muhan Li","Hai Son Le","Mahdi Bashari","Ebrahim Bagheri"],"authors_zh":"Sajad Ebrahimi、Soroush Sadeghian、Ali Ghorbanpour、Negar Arabzadeh、Sara Salamat、Muhan Li、Hai Son Le、Mahdi Bashari、Ebrahim Bagheri","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["RottenReviews benchmarks peer-review quality with human expert labels, review metadata, and LLM-based judgments.","将评审文本、学者画像与专家标注结合，量化检验同行评审质量及 LLM 裁判的可靠性。"],"why":"RottenReviews benchmarks peer-review quality with human expert labels, review metadata, and LLM-based judgments.","primary_link":"https://doi.org/10.1145/3746252.3761506","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Reviewerly-Inc/RottenReviews"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Reviewerly/RottenReviews"}],"link_count":3,"sections":9},{"id":"rpm-mcts-2025","title":"RPM-MCTS: Knowledge-Retrieval as Process Reward Model with Monte Carlo Tree Search for Code Generation","year":2025,"venue":"AAAI 2026","authors":["Yuanyuan Lin","Xiangyu Ouyang","Teng Zhang","Kaixin Sui"],"authors_zh":"Yuanyuan Lin, Xiangyu Ouyang, Teng Zhang, Kaixin Sui","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","programmatic_code_formal"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["RPM-MCTS: Knowledge-Retrieval as Process Reward Model with Monte Carlo Tree Search for Code Generation records thought/tree nodes retrieval scores similarity filter and sandbox feedback under retrieval-as-PRM and execution feedback.","RPM-MCTS 将检索相似度、沙盒反馈与 LLM 判断结合为代码规划树的过程信号。"],"why":"It makes MCTS expansion/filtering/correction and its audit boundary visible for reasoning-data curation.","primary_link":"https://arxiv.org/abs/2511.19895","links":[],"link_count":2,"sections":9},{"id":"rstar-coder-2025","title":"rStar-Coder: Scaling Competitive Code Reasoning with a Large-Scale Verified Dataset","year":2025,"venue":"NeurIPS 2025","authors":["Yifei Liu","Li Lyna Zhang","Yi Zhu","Bingcheng Dong","Xudong Zhou","Ning Shang","Fan Yang","Cheng Li","Mao Yang"],"authors_zh":"Yifei Liu、Li Lyna Zhang、Yi Zhu、Bingcheng Dong、Xudong Zhou、Ning Shang、Fan Yang、Cheng Li、Mao Yang","tracks":["data_construction_open_release_recipes","instruction_demonstration_rationale_data","programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["code","competitive-programming","algorithmic-reasoning"],"tags":["code-reasoning","competitive-programming","synthetic-data","long-cot","testcase-generation","mutual-verification","execution-verifier","rejection-sampling","rlvr","data-scaling"],"status":"partial","priority":"必读","paper_type_zh":"可验证代码推理数据发布与构造配方","best_for_zh":"构建或审计竞赛编程 SFT、蒸馏和 RLVR 数据，以及研究测试生成与一致性验证边界的读者","confidence":"medium","one_line":["rStar-Coder expands reference-bearing competition problems into synthesized problems, executable tests, and long-reasoning solutions through LLM generation, program execution, agreement filtering, and decontamination, while its public configs and supplemental code do not fully reconstruct the paper run.","rStar-Coder 将竞赛编程种子扩展为合成题、多尺度测试和长推理解答，并以程序执行和跨解答一致性筛选；但公开配置与补充代码尚不能完整重建论文运行。"],"why":"It makes problem synthesis, testcase generation, executable verification, SFT traces, and RL-ready exports visible as separate data contracts, and it provides calibrated evidence that agreement is useful but weaker than a trusted oracle.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/54e847e1dffc87a8063844b149148557-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://papers.nips.cc/paper_files/paper/2025/file/54e847e1dffc87a8063844b149148557-Supplemental-Conference.zip"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/microsoft/rStar-Coder"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/microsoft/rStar"}],"link_count":6,"sections":9},{"id":"rstar-math-2025","title":"rStar-Math: Small LLMs Can Master Math Reasoning with Self-Evolved Deep Thinking","year":2025,"venue":"ICML 2025","authors":["Xinyu Guan","Li Lyna Zhang","Yifei Liu","Ning Shang","Youran Sun","Yi Zhu","Fan Yang","Mao Yang"],"authors_zh":"Xinyu Guan、Li Lyna Zhang、Yifei Liu、Ning Shang、Youran Sun、Yi Zhu、Fan Yang、Mao Yang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","process_supervision","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","step_level","pairwise_preference","process_reward","trajectory_value"],"training_use":["sft","preference_learning","reward_modeling","process_supervision","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report"],"domains":["math","competition-mathematics"],"tags":["primary-link-checked","artifact-verified","math","code-augmented-reasoning","mcts","self-evolution","process-preference-model","process-supervision","synthetic-data","open-data"],"status":"partial","priority":"必读","paper_type_zh":"数学推理数据构造、过程偏好建模与开放数据发布","best_for_zh":"设计搜索生成的数学 SFT/过程偏好数据、复现 policy–verifier 共演化流程，或审计扁平发布造成 lineage 丢失的研究者","confidence":"high","one_line":["rStar-Math turns answer-keyed math problems into code-executed MCTS trajectories and Q-ranked step preferences across four self-evolution rounds, releasing flattened SFT and PPM tables but not the underlying source/tree lineage.","rStar-Math 通过四轮自演化，把带答案的数学题转化为经 Python 执行筛选的 MCTS 轨迹与按 Q-value 构造的步骤偏好；公开 SFT 和 PPM 扁平表，但未公开支撑审计的来源与搜索树 lineage。"],"why":"It is an operational example of jointly evolving a reasoning-data generator and process verifier without superior-model solution distillation, and it shows exactly which provenance is lost when rich search trees are flattened into training-ready SFT and preference rows.","primary_link":"https://proceedings.mlr.press/v267/guan25f.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/rStar/tree/rStar-math"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ElonTusk2001/rstar_sft"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/ElonTusk2001/rstar_ppm"}],"link_count":6,"sections":9},{"id":"rubrics-as-rewards-beyond-verifiable-domains-2025","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","year":2025,"venue":"ICLR 2026","authors":["Anisha Gunjal","Anthony Wang","Elaine Lau","Vaskar Nath","Yunzhong He","Bing Liu","Sean Hendryx"],"authors_zh":"Anisha Gunjal, Anthony Wang, Elaine Lau, Vaskar Nath, Yunzhong He, Bing Liu, Sean Hendryx","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["reward_modeling","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["scientific-reasoning","medical-reasoning","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"基于评分准则的强化学习与开放反馈数据集论文","best_for_zh":"需要将多准则、非二元评价转为可审计强化学习奖励，并在医学或科学任务中复用的研究者。","confidence":"high","one_line":["RaR trains policies with instance-specific, checklist-style rubrics, releasing medicine and science data for structured reward learning beyond binary verification.","22.9K 科学推理题含 5–12 条带权 rubric，以实例级标准作 GRPO 奖励，覆盖多类科学推理。"],"why":"The release makes the criteria behind a non-verifiable reward explicit enough to inspect, aggregate, and reuse.","primary_link":"https://openreview.net/forum?id=c1bTcrDmt4","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScaleAI/RaR-Science"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/ScaleAI/rar"},{"key":"project","label":["Project","项目主页"],"url":"https://labs.scale.com/papers/rubrics_as_rewards"}],"link_count":6,"sections":9},{"id":"rubriks-cube-explanation-rubric-cube-2025","title":"Rubrik’s Cube: Testing a New Rubric for Evaluating Explanations on the CUBE dataset","year":2025,"venue":"ACL 2025","authors":["Diana Galvan-Sosa","Gabrielle Gaudeau","Pride Kavumba","Yunmeng Li","Hongyi Gu","Zheng Yuan","Keisuke Sakaguchi","Paula Buttery"],"authors_zh":"Diana Galvan-Sosa、Gabrielle Gaudeau、Pride Kavumba、Yunmeng Li、Hongyi Gu、Zheng Yuan、Keisuke Sakaguchi、Paula Buttery","tracks":["preference_reward_feedback_data","judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning","language-understanding","explanation"],"tags":["explanations","rubric","reward-model","acl-2025"],"status":"verified","priority":"可读","paper_type_zh":"解释质量量规、标注数据集与评测研究","best_for_zh":"需要训练或评测解释质量判别器、可解释性反馈模型的研究者","confidence":"high","one_line":["Rubrik’s CUBE releases 26,000 explanations with human and model quality annotations under an education-inspired evaluation rubric.","CUBE 含约 2.6 万解释及人类、六类模型的量规质量标注，覆盖推理和语言任务，可训练可解释的质量评判器。"],"why":"It exposes rubric-conditioned explanation-quality labels rather than a single undifferentiated preference signal.","primary_link":"https://aclanthology.org/2025.acl-long.1160/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RubriksCube/rubriks_cube"}],"link_count":3,"sections":9},{"id":"s-star-code-scaling-2025","title":"S*: Test Time Scaling for Code Generation","year":2025,"venue":"Findings of EMNLP 2025","authors":["Dacheng Li","Shiyi Cao","Chengkun Cao","Xiuyu Li","Shangyin Tan","Kurt Keutzer","Jiarong Xing","Joseph E. Gonzalez","Ion Stoica"],"authors_zh":"Dacheng Li、Shiyi Cao、Chengkun Cao、Xiuyu Li、Shangyin Tan、Kurt Keutzer、Jiarong Xing、Joseph E. Gonzalez、Ion Stoica（机构以官方论文为准）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["test_time_compute"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["reasoning","software-engineering"],"tags":["test-time-compute","reasoning","scaling"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展研究（Findings of EMNLP 2025）","best_for_zh":"研究推理预算分配与测试时扩展的读者。","confidence":"high","one_line":["S*: Test Time Scaling for Code Generation","代码生成既需要更广候选覆盖，也需要可靠选择，但仅靠并行采样可能错过修复机会，而代码正确性需要可执行的区分。"],"why":"It makes a test-time decision auditable.","primary_link":"https://aclanthology.org/2025.findings-emnlp.865/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NovaSky-AI/SkyThought"}],"link_count":2,"sections":9},{"id":"s1-simple-test-time-scaling-2025","title":"s1: Simple test-time scaling","year":2025,"venue":"EMNLP 2025","authors":["Niklas Muennighoff","Zitong Yang","Weijia Shi","Xiang Lisa Li","Li Fei-Fei","Hannaneh Hajishirzi","Luke Zettlemoyer","Percy Liang","Emmanuel Candès","Tatsunori Hashimoto"],"authors_zh":"Niklas Muennighoff、Zitong Yang、Weijia Shi、Xiang Lisa Li、Li Fei-Fei、Hannaneh Hajishirzi、Luke Zettlemoyer、Percy Liang、Emmanuel Candès、Tatsunori Hashimoto","tracks":["data_construction_open_release_recipes","programmatically_verifiable_outcome_data","instruction_demonstration_rationale_data"],"source_role":["model_report","data_release","construction_recipe","scaling_study"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","distillation","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","scaling_report"],"domains":["math","science","code","logic","quantitative-reasoning"],"tags":["primary-link-checked","artifact-verified","emnlp-2025","supervised-finetuning","reasoning-distillation","data-selection","sample-efficiency","test-time-scaling","budget-forcing","open-data"],"status":"partial","priority":"必读","paper_type_zh":"紧凑推理数据选择、监督蒸馏与测试时预算控制","best_for_zh":"研究小规模高选择性 SFT、judge 驱动的数据筛选、发布可复现性，或区分训练数据 recipe 与推理期控制的读者","confidence":"high","one_line":["s1 filters a 59,029-question mixed-source pool by generation quality, Qwen difficulty, Claude-labeled diversity, and trace length into the 1,000-example Gemini-distilled s1K, then combines SFT with inference-time budget forcing.","s1 从 59,029 道混合来源题目经生成成功、格式、Qwen 难度与 Claude 领域筛选得到 1,000 条 Gemini 蒸馏 s1K，再用 inference-time budget forcing 控制思考长度；但 s1K 仅 53.6% 被 judge 判对，公开 full59K 还缺 43 行。"],"why":"The work demonstrates that careful selection can match a 59K-data ablation with roughly 1/59 as many training examples, and it cleanly separates a reusable data recipe from a decoding intervention while revealing how much lineage, judgment evidence, and determinism an apparently open release can still omit.","primary_link":"https://aclanthology.org/2025.emnlp-main.1025/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/simplescaling/s1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/simplescaling/s1K"}],"link_count":13,"sections":9},{"id":"safe-retrospective-formal-verification-2025","title":"Safe: Enhancing Mathematical Reasoning in Large Language Models via Retrospective Step-aware Formal Verification","year":2025,"venue":"ACL 2025","authors":["Chengwu Liu","Ye Yuan","Yichun Yin","Yan Xu","Xin Xu","Zaoyu Chen","Yasheng Wang","Lifeng Shang","Qun Liu","Ming Zhang"],"authors_zh":"Chengwu Liu、Ye Yuan、Yichun Yin、Yan Xu、Xin Xu、Zaoyu Chen、Yasheng Wang、Lifeng Shang、Qun Liu、Ming Zhang","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","formal-verification","process-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要用形式证明替代不透明打分来评测数学推理步骤，并研究可解释过程监督的研究者。","confidence":"high","one_line":["Safe turns each mathematical reasoning step into a Lean 4 verification problem, releasing FormalStep to make process judgments interpretable and checkable.","Safe 将数学推理的每一步回溯性地形式化为 Lean 4 验证问题，并以 FormalStep 提供可检查的逐步正确性信号。"],"why":"It grounds step-level supervision in explicit formal evidence rather than only a learned reward score.","primary_link":"https://aclanthology.org/2025.acl-long.594/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/liuchengwucn/Safe"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/liuchengwu/FormalStep"}],"link_count":5,"sections":9},{"id":"safer-or-luckier-2025","title":"Safer or Luckier? LLMs as Safety Evaluators Are Not Robust to Artifacts","year":2025,"venue":"ACL 2025","authors":["Hongyu Chen","Seraphina Goldfarb-Tarrant"],"authors_zh":"Hongyu Chen，Seraphina Goldfarb-Tarrant。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"LLM 安全评估器的 artifact 鲁棒性审计","best_for_zh":"依赖 LLM 进行成对安全评测或构建 safety jury 的团队。","confidence":"medium","one_line":["Audits artifact sensitivity, self-consistency, and human alignment in safety judges.","审计 11 个安全 judge 的 artifact 偏差、重复一致性与人类对齐，显示 jury 也无法消除风险。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://aclanthology.org/2025.acl-long.970/","links":[],"link_count":2,"sections":9},{"id":"safety-large-reasoning-models-survey-2025","title":"Safety in Large Reasoning Models: A Survey","year":2025,"venue":"Findings of EMNLP 2025","authors":["Cheng Wang","Yue Liu","Baolong Bi","Duzhen Zhang","Zhong-Zhi Li","Yingwei Ma","Yufei He","Shengju Yu","Xinfeng Li","Junfeng Fang","Jiaheng Zhang","Bryan Hooi"],"authors_zh":"Cheng Wang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning-safety","large-reasoning-models"],"tags":["foundations-and-primers","reasoning-safety","emnlp-2025","survey"],"status":"verified","priority":"可读","paper_type_zh":"大推理模型安全综述","best_for_zh":"关注强推理能力带来的安全问题与防御思路的读者。","confidence":"high","one_line":["A Findings EMNLP 2025 survey of safety risks, attacks, and defenses for large reasoning models.","系统整理大推理模型的安全风险、攻击与防御。"],"why":"It makes the safety implications of stronger reasoning capabilities explicit for readers evaluating deployment.","primary_link":"https://aclanthology.org/2025.findings-emnlp.185/","links":[],"link_count":2,"sections":9},{"id":"st-bon-2025","title":"Sampling-Efficient Test-Time Scaling: Self-Estimating the Best-of-N Sampling in Early Decoding","year":2025,"venue":"NeurIPS 2025 Spotlight","authors":["Yiming Wang","Pei Zhang","Siyuan Huang","Baosong Yang","Zhuosheng Zhang","Fei Huang","Rui Wang"],"authors_zh":"Yiming Wang；Pei Zhang；Siyuan Huang；Baosong Yang；Zhuosheng Zhang；Fei Huang；Rui Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["trajectory_value"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["best-of-n","test-time-scaling","chain-of-embedding","early-selection","hidden-states"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A Best-of-N selector that replaces explicit reward scoring with hidden-state Chain-of-Embedding information.","ST-BoN 在 Best-of-N 的早期解码阶段用内部状态的一致性估计潜在优质路径，提前截断较差候选而不训练显式奖励模型。"],"why":"It demonstrates that a test-time trace record must include selector state/score definitions even when the ranking mechanism is not a traditional reward model.","primary_link":"https://arxiv.org/abs/2503.01422","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Alsace08/ST-BoN"}],"link_count":2,"sections":9},{"id":"self-certainty-bon-2025","title":"Scalable Best-of-N Selection for Large Language Models via Self-Certainty","year":2025,"venue":"NeurIPS 2025","authors":["Zhewei Kang","Xuandong Zhao","Dawn Song"],"authors_zh":"Zhewei Kang、Xuandong Zhao、Dawn Song","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["evaluation","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","scaling_report"],"domains":["mathematics","code","reasoning"],"tags":["best-of-n","self-certainty","confidence-estimation","token-distribution","borda-voting","test-time-scaling","response-selection"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["A NeurIPS 2025 Best-of-N selector that scores each sampled response from its full token distributions and optionally combines confidence rank with answer frequency through Borda voting.","这项 NeurIPS 2025 工作用完整词元分布为每个采样回答打分以做 Best-of-N 选择，并可通过 Borda 投票把置信度排名与答案频次结合。"],"why":"For rollout/search/test-time trace curation, it specifies the candidates, token-distribution scores, ranks, extraction states, votes, and selected output, while exposing the gap between intrinsic confidence and verified correctness.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/1c7eff166a8e345f664f0faa8f4e4d2e-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/backprop07/Self-Certainty"}],"link_count":6,"sections":9},{"id":"scalingrl-dynamic-selection-2025","title":"Scale Down to Speed Up: Dynamic Data Selection for Reinforcement Learning","year":2025,"venue":"Findings of EMNLP 2025","authors":["Zhuoyue Chen","Jihai Zhang","Ben Liu","Fangquan Lin","Wotao Yin"],"authors_zh":"Zhuoyue Chen, Jihai Zhang, Ben Liu, Fangquan Lin, Wotao Yin","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["rlvr"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","mathematics"],"tags":["rlvr","dynamic-data-selection","curriculum","mathematics","ppo"],"status":"verified","priority":"必读","paper_type_zh":"动态强化学习训练数据选择研究","best_for_zh":"适合设计算力受限的数学强化学习，并需要按训练阶段明确决定哪些提示应产生下一次策略更新的读者。","confidence":"high","one_line":["ScalingRL turns a 220K mathematical prompt pool into a dynamically sampled 1.5K RL curriculum using difficulty, reasoning complexity, and reward adaptability.","ScalingRL 用难度、推理复杂度和奖励适应性，把 22 万道数学题转化为动态抽取的 1,500 题强化学习课程。"],"why":"It treats data use as an optimization schedule, rather than assuming that a fixed high-quality subset remains useful throughout RL.","primary_link":"https://aclanthology.org/2025.findings-emnlp.412/","links":[],"link_count":2,"sections":9},{"id":"scale-selective-resource-allocation-2025","title":"SCALE: Selective Resource Allocation for Overcoming Performance Bottlenecks in Mathematical Test-time Scaling","year":2025,"venue":"AAAI 2026","authors":["Yang Xiao","Chunpu Xu","Ruifeng Yuan","Jiashuo Wang","Wenjie Li","Pengfei Liu"],"authors_zh":"Yang Xiao、Chunpu Xu、Ruifeng Yuan、Jiashuo Wang、Wenjie Li、Pengfei Liu","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematics","reasoning"],"tags":["mathematical-reasoning","test-time-scaling","adaptive-compute","subproblem-decomposition","difficulty-routing","qwq","step-traces","synthetic-data","open-code","open-data"],"status":"partial","priority":"可读","paper_type_zh":"数学过程轨迹发布与子问题级测试时计算分配研究","best_for_zh":"需要构建、复现或审计子问题分解、难度路由、SFT 轨迹和测试时计算归因的研究者","confidence":"high","one_line":["SCALE releases QwQ-32B mathematical traces with decomposed steps and self-assessed difficulty scores, routing each step by a threshold and filtering only final-answer matches.","SCALE 发布带子问题、难度分数和详细步骤的 QwQ-32B 数学轨迹，并按阈值选择测试时模式；它只以最终答案匹配过滤，缺少完整选择与算力审计记录。"],"why":"It exposes richer allocation metadata than whole-problem token caps and gives an SFT-ready trace schema, but missing rejected candidates, per-step routing/compute logs, immutable SFT membership, and decontamination prevent a full selection or efficiency audit.","primary_link":"https://arxiv.org/abs/2512.00466","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/XiaoYang66/DualThinking"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/YangXiao-nlp/DualThinking"}],"link_count":5,"sections":9},{"id":"caco-code-assisted-cot-2025","title":"Scaling Code-Assisted Chain-of-Thoughts and Instructions for Model Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Honglin Lin","Qizhi Pei","Zhuoshi Pan","Yu Li","Xin Gao","Juntao Li","Conghui He","Lijun Wu"],"authors_zh":"Honglin Lin, Qizhi Pei, Zhuoshi Pan, Yu Li, Xin Gao, Juntao Li, Conghui He, Lijun Wu","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report","release_audit"],"domains":["mathematical_reasoning","code","algorithmic_reasoning"],"tags":["caco","code-assisted-cot","executable-reasoning","instruction-reversal","synthetic-data","math-reasoning","codegen","multi-stage-filtering"],"status":"partial","priority":"必读","paper_type_zh":"可执行推理数据发布与合成构建流水线","best_for_zh":"研究可执行中间表示、合成推理数据扩展、SFT 语料构建与 verifier/谱系审计的读者","confidence":"high","one_line":["Caco converts four source mixtures into 146K verified seed programs, samples about 5.3M new programs, and releases 1,348,799 reversed instruction-solution rows with their answers and code after multi-stage filtering.","Caco 将四类来源数据转为 146K 条已验证种子程序，再采样约 5.3M 个程序，并在多级过滤后发布 1,348,799 条带答案与代码的反向构造指令—解答记录。"],"why":"It makes executable code a scalable intermediate data contract, while showing why program execution, model agreement, incomplete lineage, and benchmark gains must be audited separately from semantic correctness and data quality.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/3267aa172c31ae3641c32ecc6e42e5cd-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LHL3341/Caco"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LHL3341/Caco-1.3M"}],"link_count":4,"sections":9},{"id":"agentic-test-time-scaling-2025","title":"Scaling Test-time Compute for LLM Agents","year":2025,"venue":"arXiv preprint","authors":["King Zhu","Hanhao Li","Siwei Wu","Tianshun Xing","Dehua Ma","Xiangru Tang","Minghao Liu","Jian Yang","Jiaheng Liu","Yuchen Eleanor Jiang","Changwang Zhang","Chenghua Lin","Jun Wang","Ge Zhang","Wangchunshu Zhou"],"authors_zh":"King Zhu 等（机构：OPPO AI Agent Team）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["agentic-reasoning","tool-use"],"tags":["test-time-compute","agents","tool-use","reflection","listwise-verification"],"status":"verified","priority":"可读","paper_type_zh":"智能体测试时计算扩展系统研究","best_for_zh":"将推理时搜索与验证迁移到工具使用智能体的读者。","confidence":"high","one_line":["This study maps which test-time scaling mechanisms improve tool-using language agents and finds selective reflection, list-wise comparison, and diverse rollouts effective.","该研究系统比较并行采样、选择性反思、列表式验证与多智能体轨迹多样性对语言智能体测试时扩展的影响。"],"why":"It shows that test-time scaling choices interact with agent trajectories, tool actions, and result merging rather than transferring mechanically from single-answer reasoning.","primary_link":"https://arxiv.org/abs/2506.12928","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OPPO-PersonalAI/OAgents"}],"link_count":3,"sections":9},{"id":"verification-rl-test-time-scaling-2025","title":"Scaling Test-Time Compute Without Verification or RL is Suboptimal","year":2025,"venue":"arXiv preprint","authors":["Amrith Setlur","Nived Rajaraman","Sergey Levine","Aviral Kumar"],"authors_zh":"Amrith Setlur、Aviral Kumar（卡内基梅隆大学）；Nived Rajaraman、Sergey Levine（加州大学伯克利分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","reinforcement_learning","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","general-reasoning"],"tags":["test-time-scaling","verification","reinforcement-learning","scaling-law"],"status":"verified","priority":"必读","paper_type_zh":"理论与实证测试时扩展研究","best_for_zh":"需要决定应投资验证器，还是只收集合成推理轨迹的读者。","confidence":"high","one_line":["The paper argues that verifier-guided RL or search scales test-time compute better than cloning expert traces when correct reasoning is heterogeneous.","论文论证：当正确推理具有异质性时，验证器引导的强化学习或搜索比模仿专家轨迹更能扩展测试时计算。"],"why":"It identifies verification as a structural condition for sustained test-time scaling, not merely an optional evaluation component.","primary_link":"https://arxiv.org/abs/2502.12118","links":[],"link_count":1,"sections":9},{"id":"latent-recurrent-depth-tts-2025","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","year":2025,"venue":"NeurIPS 2025","authors":["Jonas Geiping","Sean McLeish","Neel Jain","John Kirchenbauer","Siddharth Singh","Brian R. Bartoldson","Bhavya Kailkhura","Abhinav Bhatele","Tom Goldstein"],"authors_zh":"Jonas Geiping、Sean McLeish、Neel Jain、John Kirchenbauer、Siddharth Singh、Brian R. Bartoldson、Bhavya Kailkhura、Abhinav Bhatele、Tom Goldstein（机构：ELLIS Institute Tübingen、马普智能系统研究所、马里兰大学、劳伦斯利弗莫尔国家实验室等）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","software-engineering","general-reasoning"],"tags":["test-time-compute","latent-reasoning","recurrence","adaptive-compute","architecture"],"status":"verified","priority":"必读","paper_type_zh":"测试时计算扩展架构研究（NeurIPS 2025）","best_for_zh":"研究隐空间推理、循环网络与非文字化测试时扩展的读者。","confidence":"high","one_line":["A depth-recurrent language model scales inference compute by iterating hidden-state reasoning rather than producing longer visible chains.","该研究以隐空间循环深度替代更长文字思维链，使模型通过更多内部迭代扩展测试时计算。"],"why":"It treats recurrent depth as an explicit test-time budget independent of visible chain-of-thought length.","primary_link":"https://arxiv.org/abs/2502.05171","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/seal-rg/recurrent-pretraining"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/tomg-group-umd/huginn-0125"}],"link_count":5,"sections":9},{"id":"scan-self-denoising-monte-carlo-annotation-2025","title":"SCAN: Self-Denoising Monte Carlo Annotation for Robust Process Reward Learning","year":2025,"venue":"NeurIPS 2025","authors":["Yuyang Ding","Xinyu Shi","Juntao Li","Xiaobo Liang","Zhaopeng Tu","Min Zhang"],"authors_zh":"Yuyang Ding, Xinyu Shi, Juntao Li, Xiaobo Liang, Zhaopeng Tu, Min Zhang","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","process-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要以低成本自动标注训练数学过程奖励模型，并分析首错邻域噪声的研究者。","confidence":"high","one_line":["SCAN-Pro releases 197K mathematical trajectories with step scores, using confidence-guided Monte Carlo denoising for robust PRM training.","SCAN-Pro 发布 19.7 万条带步骤分数的数学轨迹，以置信度引导的蒙特卡洛去噪支持稳健过程奖励训练。"],"why":"It makes the annotation noise model and the released step-score records explicit enough to reproduce a scalable PRM-training pipeline.","primary_link":"https://arxiv.org/abs/2509.16548","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yyDing1/SCAN-PRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/dyyyyyyyy/SCAN-Pro"},{"key":"project","label":["Project","项目主页"],"url":"https://scan-prm.github.io/"}],"link_count":6,"sections":9},{"id":"sciarena-scientific-literature-evaluation-2025","title":"SciArena: An Open Evaluation Platform for Non-Verifiable Scientific Literature-Grounded Tasks","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Yilun Zhao","Kaiyan Zhang","Tiansheng Hu","Sihong Wu","Ronan Le Bras","Charles McGrady","Taira Anderson","Jonathan Bragg","Joseph Chee Chang","Jesse Dodge","Matt Latzke","Yixin Liu","Xiangru Tang","Zihang Wang","Chen Zhao","Hannaneh Hajishirzi","Doug Downey","Arman Cohan"],"authors_zh":"Yilun Zhao、Kaiyan Zhang、Tiansheng Hu、Sihong Wu、Ronan Le Bras、Charles McGrady、Taira Anderson、Jonathan Bragg、Joseph Chee Chang、Jesse Dodge、Matt Latzke、Yixin Liu、Xiangru Tang、Zihang Wang、Chen Zhao、Hannaneh Hajishirzi、Doug Downey、Arman Cohan","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"可读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["SciArena collects researcher pairwise votes to evaluate literature-grounded scientific answers without one gold response.","以科研用户的对战偏好和引用记录，评测没有唯一答案的科学文献问答。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://openreview.net/forum?id=am6RR85mnc","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yale-nlp/SciArena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/yale-nlp/SciArena"},{"key":"project","label":["Project","项目主页"],"url":"https://sciarena.ai/"}],"link_count":5,"sections":9},{"id":"scienceagentbench-scientific-discovery-agents-2025","title":"ScienceAgentBench: Toward Rigorous Assessment of Language Agents for Data-Driven Scientific Discovery","year":2025,"venue":"ICLR 2025","authors":["Ziru Chen","Shijie Chen","Yuting Ning","Qianheng Zhang","Boshi Wang","Botao Yu","Yifei Li","Zeyi Liao","Chen Wei","Zitong Lu","Vishal Dey","Mingyi Xue","Frazier N. Baker","Benjamin Burns","Daniel Adu-Ampratwum","Xuhui Huang","Xia Ning","Song Gao","Yu Su","Huan Sun"],"authors_zh":"Ziru Chen、Shijie Chen、Yuting Ning、Qianheng Zhang、Boshi Wang、Botao Yu、Yifei Li、Zeyi Liao、Chen Wei、Zitong Lu、Vishal Dey、Mingyi Xue、Frazier N. Baker、Benjamin Burns、Daniel Adu-Ampratwum、Xuhui Huang、Xia Ning、Song Gao、Yu Su、Huan Sun","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["benchmark","expert_evaluation"],"tags":["benchmark","expert_evaluation","judgment"],"status":"verified","priority":"必读","paper_type_zh":"基准与评测论文","best_for_zh":"需要使用专家题目、评审或评分信号评测推理系统的研究者。","confidence":"high","one_line":["ScienceAgentBench rigorously evaluates data-tool-using language agents on expert-validated scientific discovery tasks.","以专家验证的科学发现任务，严谨评测可调用数据工具的语言 Agent。"],"why":"It makes expert-grounded evaluation evidence and its audit boundary visible.","primary_link":"https://openreview.net/pdf?id=6z4YKr0GK6","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OSU-NLP-Group/ScienceAgentBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/osunlp/ScienceAgentBench"}],"link_count":4,"sections":9},{"id":"scienceboard-2025","title":"ScienceBoard: Evaluating Multimodal Autonomous Agents in Realistic Scientific Workflows","year":2025,"venue":"ICLR 2026","authors":["Qiushi Sun","Zhoumianze Liu","Chang Ma","Zichen Ding","Fangzhi Xu","Zhangyue Yin","Haiteng Zhao","Zhenyu Wu","Kanzhi Cheng","Zhaoyang Liu","Jianing Wang","Qintong Li","Xiangru Tang","Tianbao Xie","Xiachong Feng","Xiang Li","Ben Kao","Wenhai Wang","Biqing Qi","Lingpeng Kong","Zhiyong Wu"],"authors_zh":"Qiushi Sun、Zhoumianze Liu、Chang Ma、Zichen Ding、Fangzhi Xu、Zhangyue Yin、Haiteng Zhao、Zhenyu Wu、Kanzhi Cheng、Zhaoyang Liu、Jianing Wang、Qintong Li、Xiangru Tang、Tianbao Xie、Xiachong Feng、Xiang Li、Ben Kao、Wenhai Wang、Biqing Qi、Lingpeng Kong、Zhiyong Wu","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","benchmark","verifier_reward","data_release"],"verification_contract":["environmental"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction"],"tags":["environment-agent-trajectory-data","agent-trajectories","evaluation"],"status":"partial","priority":"必读","paper_type_zh":"科学工作流 computer-use 环境、基准与轨迹发布","best_for_zh":"研究科学智能体、GUI/CLI 混合操作、环境反馈、状态验证与可重放性的读者","confidence":"medium","one_line":["ScienceBoard evaluates computer-use agents on 169 human-curated scientific workflows across six domains using a released Ubuntu VM, task-specific internal-state evaluators, and partially documented trajectory archives.","ScienceBoard 用 169 个六领域科学工作流任务、Ubuntu 科学软件 VM 和任务特定内部状态 evaluator 评测 computer-use agents，并公开环境、代码及部分轨迹压缩包。"],"why":"It exposes professional scientific applications through GUI and CLI while validating workflow I/O and application-internal state, making verifier design and environment reproducibility central audit objects.","primary_link":"https://arxiv.org/abs/2505.19897","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OS-Copilot/ScienceBoard"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OS-Copilot/ScienceBoard-Traj"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/OS-Copilot/ScienceBoard-Env"},{"key":"project","label":["Project","项目主页"],"url":"https://qiushisun.github.io/ScienceBoard-Home/"}],"link_count":7,"sections":9},{"id":"seapo-error-amplification-preference-2025","title":"SeaPO: Strategic Error Amplification for Robust Preference Optimization of Large Language Models","year":2025,"venue":"Findings of EMNLP 2025","authors":["Jun Rao","Yunjie Liao","Xuebo Liu","Zepeng Lin","Lian Lian","Dong Jin","Shengjun Cheng","Jun Yu","Min Zhang"],"authors_zh":"Jun Rao, Yunjie Liao, Xuebo Liu, Zepeng Lin, Lian Lian, Dong Jin, Shengjun Cheng, Jun Yu, Min Zhang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["reasoning","code_generation","truthfulness"],"tags":["preference-optimization","synthetic-negatives","error-injection","data-construction"],"status":"verified","priority":"必读","paper_type_zh":"合成偏好负样本构造研究","best_for_zh":"适合奖励模型排序难以产生明显负例、需要自行设计偏好数据的读者。","confidence":"high","one_line":["SeaPO creates deliberately more erroneous rejected responses so preference optimization can suppress targeted reasoning, logic, and hallucination errors.","SeaPO 刻意构造错误更明显的被拒回答，使偏好优化能压低指定的推理、逻辑和幻觉错误。"],"why":"It changes the content and severity of negative preference evidence, making the data itself encode the failure modes training should reduce.","primary_link":"https://aclanthology.org/2025.findings-emnlp.898/","links":[],"link_count":2,"sections":9},{"id":"search-o1-2025","title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","year":2025,"venue":"EMNLP 2025","authors":["Xiaoxi Li","Guanting Dong","Jiajie Jin","Yuyao Zhang","Yujia Zhou","Yutao Zhu","Peitian Zhang","Zhicheng Dou"],"authors_zh":"Xiaoxi Li, Guanting Dong, Jiajie Jin, Yuyao Zhang, Yujia Zhou, Yutao Zhu, Peitian Zhang, Zhicheng Dou","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","raw_search_rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["Search-o1 interleaves QwQ-32B-Preview reasoning with live top-10 web retrieval and a separate Reason-in-Documents rewrite, but releases no versioned trajectory corpus or web snapshot.","Search-o1 在推理时生成查询、检索网页并压缩文档内容；它是推理时搜索框架而非训练轨迹发布。"],"why":"It exposes a useful test-time trace schema and transformation boundary between raw retrieval and model-written evidence, while making replay, provenance, and rights gaps explicit.","primary_link":"https://aclanthology.org/2025.emnlp-main.276/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUC-NLPIR/Search-o1"},{"key":"project","label":["Project","项目主页"],"url":"https://search-o1.github.io/"}],"link_count":6,"sections":9},{"id":"search-time-data-contamination-2025","title":"Search-Time Data Contamination","year":2025,"venue":"NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle","authors":[],"authors_zh":"Ziwen Han, Meher Mankikar, Julian Michael, Zifan Wang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":[],"tags":["seeded-from-bib"],"status":"verified","priority":"可读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Local BibTeX seed for the 🧭 Surveys and Primers map; use it to inspect the paper's data object, verifier contract, and release metadata before promoting it.","把检索或搜索阶段引入的泄漏作为独立污染源，提示评测不能只检查训练语料重叠。"],"why":"Official paper link is pinned; curator should next add a paper-specific reasoning-data summary and audit note.","primary_link":"https://neurips.cc/virtual/2025/122466","links":[],"link_count":2,"sections":9},{"id":"seed-coder-2025","title":"Seed-Coder: Let the Code Model Curate Data for Itself","year":2025,"venue":"arXiv preprint","authors":["ByteDance Seed","Yuyu Zhang","Jing Su","Yifan Sun","Chenguang Xi","Xia Xiao","Shen Zheng","Anxiang Zhang","Kaibo Liu","Daoguang Zan","Tao Sun","Jinhua Zhu","Shulin Xin","Dong Huang","Yetao Bai","Lixin Dong","Chao Li","Jianchong Chen","Hanzhi Zhou","Yifan Huang","Guanghan Ning","Xierui Song","Jiaze Chen","Siyao Liu","Kai Shen","Liang Xiang","Yonghui Wu"],"authors_zh":"ByteDance Seed 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","distillation","preference_learning","reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["code_generation","code_completion","code_infilling","code_editing","software_engineering","competitive_programming","multilingual_code"],"tags":["seed-coder","bytedance-seed","code-data","model-centric-curation","quality-scorer","deepseek-v2-chat","synthetic-sft","generated-tests","dpo","long-cot","grpo","rlvr","competitive-programming","data-feedback-loop","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"代码模型数据构造与推理后训练技术报告","best_for_zh":"研究代码语料过滤、synthetic SFT、execution feedback、DPO、RLVR 与数据审计的读者","confidence":"high","one_line":["Seed-Coder builds a 6T curriculum with external-teacher quality scoring, then adds 3M SFT, 20K sandbox-derived preferences, and 250-step LongCoT GRPO while withholding data, scorer, tests, rewards, and lineage.","Seed-Coder 用外部 teacher 标注与 1.3B scorer 构造 6T-token curriculum，再加入 3M SFT、20K sandbox preference 与 250-step LongCoT GRPO，但未发布数据、scorer、测试、reward 或记录级 lineage。"],"why":"It shows how learned filters and executable feedback shape a code model while exposing teacher, language, formatting, generated-test, licensing, and leakage risks.","primary_link":"https://arxiv.org/abs/2506.03524","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ByteDance-Seed/Seed-Coder"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/ByteDance-Seed/seed-coder"},{"key":"project","label":["Project","项目主页"],"url":"https://bytedance-seed-coder.github.io"}],"link_count":5,"sections":9},{"id":"seed-prover-1-5-2025","title":"Seed-Prover 1.5: Mastering Undergraduate-Level Theorem Proving via Learning from Experience","year":2025,"venue":"arXiv preprint","authors":["Jiangjie Chen","Wenxiang Chen","Jiacheng Du","Jinyi Hu","Zhicheng Jiang","Allan Jie","Xiaoran Jin","Xing Jin","Chenggang Li","Wenlei Shi","Zhihong Wang","Mingxuan Wang","Chenrui Wei","Shufa Wei","Huajian Xin","Fan Yang","Weihao Gao","Zheng Yuan","Tianyang Zhan","Zeyu Zheng","Tianxi Zhou","Thomas Hanwen Zhu"],"authors_zh":"Jiangjie Chen、Wenxiang Chen、Jiacheng Du 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","agent_environment","construction_recipe","scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["formal_mathematics","lean_theorem_proving","undergraduate_mathematics","graduate_mathematics","algebra","analysis","number_theory","combinatorics","mathematical_research"],"tags":["seed-prover-1-5","bytedance-seed","doubao-seedprover","lean4","agentic-rl","tool-use","experience-trajectories","compiler-feedback","mathlib-search","self-summarization","rubric-rl","natural-language-to-lean","proof-sketch","test-time-scaling","putnambench","fate","verifier-reward","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"形式化证明 Agent 模型报告与数据披露账本","best_for_zh":"研究工具调用 RL、形式化经验轨迹、proof sketch、长程 TTS 与基准审计的读者","confidence":"high","one_line":["Seed-Prover 1.5 turns Lean, Mathlib search, Python calls, cached lemmas, and failure summaries into outcome-rewarded trajectories, then adds rubric-trained NL-to-Lean sketches and recursive proof search; 11 Putnam 2025 proofs are open, but training experience is not.","Seed-Prover 1.5 将 Lean、Mathlib 搜索、Python 调用、缓存 lemma 与失败摘要组织为终止奖励的 agent 轨迹，再以 Rubric RL 的 NL-to-Lean sketch 和递归证明树扩展推理；公开的 11 个 Putnam 2025 证明并不等于训练经验数据发布。"],"why":"It concretizes tool-use theorem-agent experience and natural-language-to-formal decomposition while showing why final Lean validity cannot settle contamination, judge-shaped decomposition, formalization fidelity, tool dependence, or compute comparability.","primary_link":"https://arxiv.org/abs/2512.17260","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ByteDance-Seed/Seed-Prover"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ByteDance-Seed/Seed-Prover/tree/main/SeedProver-1.5"}],"link_count":4,"sections":9},{"id":"seed-prover-2025","title":"Seed-Prover: Deep and Broad Reasoning for Automated Theorem Proving","year":2025,"venue":"arXiv preprint","authors":["Luoxin Chen","Jinming Gu","Liankai Huang","Wenhao Huang","Zhicheng Jiang","Allan Jie","Xiaoran Jin","Xing Jin","Chenggang Li","Kaijing Ma","Cheng Ren","Jiawei Shen","Wenlei Shi","Tong Sun","He Sun","Jiahui Wang","Siran Wang","Zhihong Wang","Chenrui Wei","Shufa Wei","Yonghui Wu","Yuchen Wu","Yihang Xia","Huajian Xin","Fan Yang","Huaiyuan Ying","Hongyi Yuan","Zheng Yuan","Tianyang Zhan","Chi Zhang","Yue Zhang","Ge Zhang","Tianyun Zhao","Jianqiu Zhao","Yichi Zhou","Thomas Hanwen Zhu"],"authors_zh":"Luoxin Chen、Jinming Gu、Liankai Huang 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","verifier_reward","construction_recipe","scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["formal_mathematics","lean_theorem_proving","olympiad_mathematics","geometry","algebra","number_theory","combinatorics"],"tags":["seed-prover","bytedance-seed","lean4","formal-theorem-proving","lemma-style-proving","compiler-feedback","self-summarization","conjecture-pool","lemma-pool","vapo","rlvr","test-time-scaling","seed-geometry","synthetic-geometry","verifier-reward","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"形式化定理证明模型报告与数据披露账本","best_for_zh":"研究 Lean RLVR、失败轨迹与编译反馈、测试时搜索扩展和形式化数据审计的读者","confidence":"high","one_line":["Seed-Prover trains a lemma-first Lean policy with binary proof rewards and feedback-rich prompts, then scales from iterative repair to 5,000-conjecture search; Seed-Geometry adds a reported 230M-problem corpus, while only successful proof artifacts are open.","Seed-Prover 以 Lean 二值成功奖励训练 lemma-first 证明策略，并从编译反馈迭代扩展到 5,000 个猜想的重搜索；Seed-Geometry 报告了 2.3 亿问题，但公开仓库仅含成功证明 artifact，不是训练数据发布。"],"why":"It makes the reasoning-data object concrete—formal statements, lemmas, dependencies, failures, compiler feedback, summaries, conjectures, and verified terminal reward—while exposing missing membership, failed trajectories, judge-assisted selection, and search-compute accounting.","primary_link":"https://arxiv.org/abs/2507.23726","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ByteDance-Seed/Seed-Prover"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ByteDance-Seed/Seed-Prover/tree/main/SeedProver"}],"link_count":4,"sections":9},{"id":"seed1-5-thinking-2025","title":"Seed1.5-Thinking: Advancing Superb Reasoning Models with Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["ByteDance Seed"],"authors_zh":"ByteDance Seed","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["mathematics","physics","chemistry","coding","logic_reasoning","general_assistant"],"tags":["frontier-report","data-disclosure-ledger","long-cot","sft","rlvr","verifier","reward-modeling"],"status":"partial","priority":"必读","paper_type_zh":"前沿推理模型报告与数据披露账本","best_for_zh":"希望核对推理 RL 数据来源、反馈接口、训练配方与审计缺口的研究者","confidence":"high","one_line":["ByteDance Seed reports four reasoning-RL data families, 400k long-CoT SFT examples, filters, verifiers, and reward modeling without releasing source-level data or full audit metadata.","ByteDance Seed 披露四类推理 RL 数据、40 万条 long-CoT SFT、过滤器、verifier 与 reward modeling，但未发布来源级数据或完整审计元数据。"],"why":"Lets readers separate disclosed data and feedback interfaces from undisclosed provenance, licensing, split, decontamination, and reusable artifacts.","primary_link":"https://arxiv.org/abs/2504.13914","links":[],"link_count":2,"sections":9},{"id":"seed1-5-vl-technical-report-2025","title":"Seed1.5-VL Technical Report","year":2025,"venue":"arXiv preprint","authors":["ByteDance Seed Team"],"authors_zh":"ByteDance Seed Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","preference_learning","rlvr"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["multimodal","general_reasoning"],"tags":["bytedance-seed","seed1-5-vl","frontier-report","data-disclosure-ledger","multimodal","sft","rlhf","rlvr"],"status":"partial","priority":"必读","paper_type_zh":"多模态前沿模型技术报告","best_for_zh":"审计多模态后训练中数据与反馈披露边界的读者","confidence":"high","one_line":["Seed1.5-VL discloses a 3T-scale multimodal pipeline and hybrid RLHF/RLVR recipe, while releasing API access and sample code rather than model weights or training records.","Seed1.5-VL 报告了彼此分离的数据构建、SFT、RLHF 与 RLVR 阶段，但未确立可复用数据工件或完整、可审计的反馈配方。"],"why":"It is a high-information disclosure ledger for multimodal post-training because it connects images, prompts, LongCoT traces, preference rewards and rule verifiers without making the hidden data reusable.","primary_link":"https://arxiv.org/abs/2505.07062","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ByteDance-Seed/Seed1.5-VL"},{"key":"project","label":["Project","项目主页"],"url":"https://seed.bytedance.com/en/tech/seed1_5_vl"}],"link_count":5,"sections":9},{"id":"sepo-token-selection-2025","title":"Selective Preference Optimization via Token-Level Reward Function Estimation","year":2025,"venue":"EMNLP 2025 Main Conference","authors":["Kailai Yang","Zhiwei Liu","Qianqian Xie","Jimin Huang","Erxue Min","Sophia Ananiadou"],"authors_zh":"Kailai Yang, Zhiwei Liu, Qianqian Xie, Jimin Huang, Erxue Min, Sophia Ananiadou","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["instruction-tuning","alignment"],"tags":["preference-optimization","token-selection","dpo","weak-to-strong","training-efficiency"],"status":"verified","priority":"必读","paper_type_zh":"token 级选择性偏好优化研究","best_for_zh":"希望从既有成对数据中选择 token 级监督、降低偏好训练成本或噪声的读者。","confidence":"high","one_line":["SePO trains a small DPO oracle to select only high-value chosen tokens and low-value rejected tokens for preference optimization.","SePO 训练一个小型 DPO oracle，从偏好对中选出高价值的入选 token 与低价值的拒绝 token 进行偏好优化。"],"why":"It turns response-level preferences into an explicit token mask that determines which supervision signals reach the target policy.","primary_link":"https://aclanthology.org/2025.emnlp-main.359/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SteveKGYang/SePO"}],"link_count":4,"sections":9},{"id":"self-consistency-preference-optimization-2025","title":"Self-Consistency Preference Optimization","year":2025,"venue":"ICML 2025","authors":["Archiki Prasad","Weizhe Yuan","Richard Yuanzhe Pang","Jing Xu","Maryam Fazel-Zarandi","Mohit Bansal","Sainbayar Sukhbaatar","Jason E. Weston","Jane Yu"],"authors_zh":"Archiki Prasad；Weizhe Yuan；Richard Yuanzhe Pang；Jing Xu；Maryam Fazel-Zarandi；Mohit Bansal；Sainbayar Sukhbaatar；Jason E. Weston；Jane Yu","tracks":["rollout_search_test_time_trace_data","training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["preference_learning","evaluation"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["mathematics","reasoning"],"tags":["self-consistency","preference-optimization","dpo","majority-vote","rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A DPO construction recipe that derives preference labels from repeated-answer self-consistency.","ScPO 对同一问题采样多份答案，依据答案频次把更一致的回答作为偏好对的正例，用 DPO 训练模型偏向一致答案。"],"why":"It makes a key Track 5 conversion explicit—rollout counts become preference data—but requires raw samples and split discipline to audit leakage.","primary_link":"https://proceedings.mlr.press/v267/prasad25a.html","links":[],"link_count":3,"sections":9},{"id":"self-rewarding-correction-math-2025","title":"Self-rewarding correction for mathematical reasoning","year":2025,"venue":"arXiv","authors":["Wei Xiong","Hanning Zhang","Chenlu Ye","Lichang Chen","Nan Jiang","Tong Zhang"],"authors_zh":"Wei Xiong、Hanning Zhang、Chenlu Ye、Lichang Chen、Nan Jiang、Tong Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","verifier_reward","model_report"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level","scalar_reward","pairwise_preference"],"training_use":["sft","preference_learning","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["math","reasoning","self_correction"],"tags":["self-rewarding","self-correction","sequential-rejection-sampling","mathematical-reasoning","generative-verifier","ppo","multi-turn-dpo","open-release"],"status":"partial","priority":"必读","paper_type_zh":"数学自纠正轨迹构造与强化学习配方","best_for_zh":"关注自评、自纠正、拒绝采样和多阶段后训练的研究者","confidence":"high","one_line":["Sequential rejection sampling builds verifier-selected answer/evaluation/correction trajectories for SFT before correctness-reward PPO or DPO; the 31,990-row final release lacks complete construction lineage.","公开 31990 条经筛选的自评/纠正轨迹及 SFT 到 PPO/DPO 配方，但谱系、拒绝数和许可证不完整。"],"why":"For construction-recipe readers, it makes the control tokens, ground-truth filter, SFT object, and downstream RL contracts separable, while showing why accepted traces and benchmark gains do not substitute for candidate, rejection, license, and contamination records.","primary_link":"https://arxiv.org/abs/2502.19613","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RLHFlow/Self-rewarding-reasoning-LLM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/RLHFlow/self_rewarding_ift_example"}],"link_count":4,"sections":9},{"id":"self-steering-preference-optimization-2025","title":"Self-Steering Optimization: Autonomous Preference Optimization for Large Language Models","year":2025,"venue":"Findings of ACL 2025","authors":["Hao Xiang","Bowen Yu","Hongyu Lin","Keming Lu","Yaojie Lu","Xianpei Han","Ben He","Le Sun","Jingren Zhou","Junyang Lin"],"authors_zh":"Hao Xiang, Bowen Yu, Hongyu Lin, Keming Lu, Yaojie Lu, Xianpei Han, Ben He, Le Sun, Jingren Zhou, Junyang Lin","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","preference_data","reasoning"],"tags":["automated-alignment","preference-data","dpo","on-policy","principle-conditioning"],"status":"verified","priority":"必读","paper_type_zh":"自动偏好数据生成研究","best_for_zh":"适合设计自生成偏好数据、并需要同时控制回答对正确性和被拒绝回答在目标策略下概率的读者。","confidence":"high","one_line":["Self-Steering Optimization trains a policy-derived generator to synthesize accurate, near-on-policy preference pairs for later preference optimization.","Self-Steering Optimization 训练一个源自策略模型的生成器，为后续偏好优化合成准确且接近 on-policy 的偏好对。"],"why":"It makes response-distribution control part of the preference-record construction stage instead of assuming that principle prompting alone yields trainable pairs.","primary_link":"https://aclanthology.org/2025.findings-acl.473/","links":[],"link_count":2,"sections":9},{"id":"self-training-elicits-concise-reasoning-2025","title":"Self-Training Elicits Concise Reasoning in Large Language Models","year":2025,"venue":"Findings of the Association for Computational Linguistics: ACL 2025","authors":["Tergel Munkhbat","Namgyu Ho","Seo Hyun Kim","Yongjin Yang","Yujin Kim","Se-Young Yun"],"authors_zh":"Tergel Munkhbat、Namgyu Ho、Seo Hyun Kim、Yongjin Yang、Yujin Kim、Se-Young Yun","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer"],"domains":["mathematical_reasoning"],"tags":["best-of-n","self-training","concise-reasoning","shortest-correct-selection","few-shot-conditioning","gsm8k","math","sft","test-time-compute"],"status":"partial","priority":"可读","paper_type_zh":"自训练推理轨迹构造与简洁推理微调方法","best_for_zh":"关注训练时搜索、best-of-N 轨迹选择、答案验证和推理 token 成本的读者","confidence":"high","one_line":["Per-question self-training selects the shortest parser-correct target-model rollout—optionally from few-shot-conditioned candidates—and distills it with one-epoch SFT for shorter greedy math reasoning.","该方法从目标模型在数学题上生成的多个推理路径中选择最终答案经解析器核验的最短路径（可结合少样本条件），再用于 SFT 以获得更简洁的推理。"],"why":"It converts best-of-N search into concise SFT traces for the rollout-search track, while exposing the need to audit training-time compute, parser errors, dropped hard questions, and the unreleased accepted/rejected trajectory ledger.","primary_link":"https://aclanthology.org/2025.findings-acl.1289/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TergelMunkhbat/concise-reasoning"}],"link_count":4,"sections":9},{"id":"crest-consistency-rationale-self-training-2025","title":"Self-Training Meets Consistency: Improving LLMs' Reasoning with Consistency-Driven Rationale Evaluation","year":2025,"venue":"NAACL 2025","authors":["Jaehyeok Lee","Keisuke Sakaguchi","JinYeong Bak"],"authors_zh":"Jaehyeok Lee、Keisuke Sakaguchi、JinYeong Bak","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["logical_reasoning","commonsense_reasoning","science_reasoning"],"tags":["self-training","rationale-filtering","option-wise-consistency","mixed-preferences","dpo","same-model-evaluation","missing-derived-data"],"status":"partial","priority":"可读","paper_type_zh":"一致性驱动的 rationale 自训练配方","best_for_zh":"关注 rationale 筛选、自训练、偏好构造和同模型评估偏差的研究者","confidence":"medium","one_line":["CREST samples 16 rationales per multiple-choice item, scores originally correct candidates with option-wise consistency probes, and converts the scores into tolerance-filtered SFT records and mixed DPO pairs.","按原题答案和选项级追问一致性评价每题 16 条自生成 rationale，再构造容错 SFT 与混合 DPO 数据。"],"why":"It makes a denser label-derived rationale-selection recipe inspectable while showing that same-model probing, aggregate counts and an unreleased decision ledger limit auditability.","primary_link":"https://aclanthology.org/2025.naacl-long.528/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JaehyeokLee-119/CREST"},{"key":"data","label":["Data","数据"],"url":"https://github.com/JaehyeokLee-119/CREST/tree/main/resources/data"}],"link_count":6,"sections":9},{"id":"serl-self-play-reinforcement-learning-2025","title":"SeRL: Self-play Reinforcement Learning for Large Language Models with Limited Data","year":2025,"venue":"NeurIPS 2025 (Main Conference Track)","authors":["Wenkai Fang","Shunyu Liu","Yang Zhou","Kongcheng Zhang","Tongya Zheng","Kaixuan Chen","Mingli Song","Dacheng Tao"],"authors_zh":"Wenkai Fang、Shunyu Liu、Yang Zhou、Kongcheng Zhang、Tongya Zheng、Kaixuan Chen、Mingli Song、Dacheng Tao","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","data_release","model_report"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr"],"construction_layer":["prompt_sourcing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning","medical_reasoning","self_play"],"tags":["serl","self-play","self-instruction","self-rewarding","majority-voting","math-verify","reinforce-plus-plus","online-data-generation","limited-data","reward-hacking"],"status":"partial","priority":"必读","paper_type_zh":"有限 seed 数据的在线 self-play RLVR 构造 recipe、programmatic self-reward 与部分数据发布","best_for_zh":"研究 self-instruction、self-consensus reward、online curriculum、reward hacking，或审计 paper-code-data 一致性与 trajectory release 完整性的读者","confidence":"high","one_line":["SeRL generates questions from a 500-example seed, filters them by diversity and self-consensus difficulty, and trains on sixteen-response agreement rewards, while releasing code and prompt snapshots rather than complete paper-run trajectories.","SeRL 从 500 条 seed instructions 出发，在线生成问题，以 16 个响应的 Math-Verify 等价聚类和多数一致构造二元奖励，再用 Reinforce++ 更新同一 policy；但公开物只有代码与 prompt snapshots，缺少论文运行的完整 responses、rewards、failures、logs、checkpoints 与 manifests，并存在 paper/code 配置漂移和 2,802 条 exact duplicate prompts。"],"why":"It operationalizes label-free reward construction for scarce-domain reasoning, but its wrong-consensus failure and paper-code-release mismatches show why self-rewarding data needs failures, equivalence traces, provenance, and immutable configurations.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/file/95c2cbe23fb6d28a4ae908aa7f3de5bf-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wantbook-book/SeRL"},{"key":"data","label":["Data","数据"],"url":"https://github.com/wantbook-book/SeRL/tree/f8a4f86a797b9a91b22ce53b1fdc7fea4c225693/openrlhf/dataset/math"}],"link_count":8,"sections":9},{"id":"sets-self-verification-self-correction-2025","title":"SETS: Leveraging Self-Verification and Self-Correction for Improved Test-Time Scaling","year":2025,"venue":"TMLR","authors":["Jiefeng Chen","Jie Ren","Xinyun Chen","Chengrun Yang","Ruoxi Sun","Jinsung Yoon","Sercan Ö. Arık"],"authors_zh":"Jiefeng Chen；Jie Ren；Xinyun Chen；Chengrun Yang；Ruoxi Sun；Jinsung Yoon；Sercan Ö. Arık","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["self-verification","self-correction","test-time-scaling","exact-mode","multi-sample"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A training-free self-verification/self-correction loop with explicit repeated-answer and correction objects.","SETS 先并行生成多个答案，再由模型自我核验并在有限轮次内自我修正；它依赖模型自身判断，不等同于外部正确性验证器。"],"why":"It is useful for separating test-time traces from training data and for documenting why self-judged correctness needs evidence beyond final accuracy.","primary_link":"https://arxiv.org/abs/2501.19306","links":[],"link_count":1,"sections":9},{"id":"sirius-self-improving-multi-agent-systems-2025","title":"SiriuS: Self-improving Multi-agent Systems via Bootstrapped Reasoning","year":2025,"venue":"NeurIPS 2025 (Main Conference Track)","authors":["Wanjia Zhao","Mert Yuksekgonul","Shirley Wu","James Y. Zou"],"authors_zh":"Wanjia Zhao、Mert Yuksekgonul、Shirley Wu、James Y. Zou","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","agent_environment","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["multi_agent_reasoning","scientific_reasoning","biomedical_qa","negotiation"],"tags":["sirius","multi-agent","bootstrapped-reasoning","experience-library","trajectory-augmentation","role-specific-sft","outcome-supervision","negotiation","release-incomplete"],"status":"partial","priority":"必读","paper_type_zh":"多智能体 outcome-supervised 数据构建与失败轨迹修复配方","best_for_zh":"研究角色级 SFT、终局 credit propagation、feedback repair、Actor-Critic 与 negotiation，以及 experience-library 发布和污染审计","confidence":"high","one_line":["SiriuS turns terminally successful interactions into per-role SFT records and repairs failed QA episodes through critique, regeneration, rephrasing, and downstream replay, but does not release the libraries.","SiriuS 用终局成功筛选角色级 SFT 记录，并通过 ground-truth-guided critique、regeneration、rephrasing 与下游 replay 修复失败轨迹；当前只公开五条 physics sample input，未公开论文 experience library、失败账本、feedback、模型或日志。"],"why":"It operationalizes outcome-supervised multi-agent self-improvement while exposing coarse credit assignment and absent failure/library provenance.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/file/b45279ac82cb017a5f55ea7d3653193a-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zou-group/sirius"},{"key":"data","label":["Data","数据"],"url":"https://github.com/zou-group/sirius/blob/16643cdc484b07d4d20419ba785a32a9845639b7/dataset/phy_train.jsonl"}],"link_count":10,"sections":9},{"id":"skip-thinking-2025","title":"Skip-Thinking: Chunk-wise Chain-of-Thought Distillation Enable Smaller Language Models to Reason Better and Faster","year":2025,"venue":"EMNLP 2025","authors":["Xiaoshu Chen","Sihang Zhou","Ke Liang","Xiaoyu Sun","Xinwang Liu"],"authors_zh":"Xiaoshu Chen；Sihang Zhou；Ke Liang；Xiaoyu Sun；Xinwang Liu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["distillation","evaluation"],"construction_layer":["trace_writing","optimizer_scaffold"],"domains":["reasoning"],"tags":["cot-compression","selective-skipping","distillation","teacher-traces","efficiency"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A selective CoT compression approach that treats omitted intermediate steps as a controlled construction decision.","Skip-Thinking 以 text-davinci-002 生成的 CoT 为教师轨迹，按 chunk 进行 CWT、SBC 与 STT 等选择性跳过，以训练更短的推理。"],"why":"It helps audit long-to-short trace transformations, especially the need to retain chunking and deletion decisions alongside resulting text.","primary_link":"https://aclanthology.org/2025.emnlp-main.610/","links":[],"link_count":2,"sections":9},{"id":"skywork-reward-v2-synpref-40m-2025","title":"Skywork-Reward-V2: Scaling Preference Data Curation via Human-AI Synergy","year":2025,"venue":"arXiv","authors":["Chris Yuhao Liu","Liang Zeng","Yuzhen Xiao","Jujie He et al."],"authors_zh":"Chris Yuhao Liu、Liang Zeng、Yuzhen Xiao、Jujie He 等","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["preference-feedback-batch-2026","post-training","data-construction"],"status":"verified","priority":"暂缓","paper_type_zh":"偏好与奖励反馈数据集或数据构建研究","best_for_zh":"需要构建、审计或复用偏好与奖励反馈数据的研究者。","confidence":"high","one_line":["Skywork-Reward-V2: Scaling Preference Data Curation via Human-AI Synergy contributes a preference/reward feedback data object or construction method.","发布 SynPref-40M 并提出人机协同的数据筛选与校验流程，用大规模高质量偏好数据训练通用奖励模型。"],"why":"It exposes a reusable preference or reward-feedback data surface that requires provenance and bias audit before reuse.","primary_link":"https://arxiv.org/abs/2507.01352","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SkyworkAI/Skywork-Reward-V2"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/Skywork/skywork-reward-v2-6867a90eaf509dd6e591e4c1"}],"link_count":3,"sections":9},{"id":"smoltalk-2025","title":"SmolLM2: When Smol Goes Big - Data-Centric Training of a Small Language Model","year":2025,"venue":"arXiv preprint (2025)","authors":["Loubna Ben Allal","Anton Lozhkov","Elie Bakouch","Gabriel Martín Blázquez","Guilherme Penedo","Lewis Tunstall","Andrés Marafioti","Hynek Kydlíček","Agustín Piqueres Lajarín","Vaibhav Srivastav","Joshua Lochner","Caleb Fahlgren","Xuan-Son Nguyen","Clémentine Fourrier","Ben Burtenshaw","Hugo Larcher","Haojun Zhao","Cyril Zakka","Mathieu Morlon","Colin Raffel","Leandro von Werra","Thomas Wolf"],"authors_zh":"Loubna Ben Allal、Anton Lozhkov、Elie Bakouch、Gabriel Martín Blázquez、Guilherme Penedo、Lewis Tunstall、Andrés Marafioti、Hynek Kydlíček、Agustín Piqueres Lajarín、Vaibhav Srivastav、Joshua Lochner、Caleb Fahlgren、Xuan-Son Nguyen、Clémentine Fourrier、Ben Burtenshaw、Hugo Larcher、Haojun Zhao、Cyril Zakka、Mathieu Morlon、Colin Raffel、Leandro von Werra、Thomas Wolf","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["primarily English chat, mathematics, code, rewriting, constraints, and everyday instruction following"],"tags":["instruction-demonstration-rationale","arxiv-2502.02737","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"SmolLM2 的指令监督微调","confidence":"high","one_line":["SmolTalk combines inherited and newly generated conversations under one message schema, then tunes source weights through ablations and manual mixture review.","SmolTalk 用统一对话结构汇集约 110 万条指令记录，并通过消融与人工复核调整小模型的后训练配比。"],"why":"Small language models need broad instruction coverage, but existing open chat mixtures were too small or uneven for the paper's compact-model training budget.","primary_link":"https://arxiv.org/abs/2502.02737","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HuggingFaceTB/smoltalk"}],"link_count":2,"sections":9},{"id":"socratic-mcts-2025","title":"Socratic-MCTS: Test-Time Visual Reasoning by Asking the Right Questions","year":2025,"venue":"EMNLP 2025","authors":["David Acuna","Ximing Lu","Jaehun Jung","Hyunwoo Kim","Amlan Kar","Sanja Fidler","Yejin Choi"],"authors_zh":"David Acuna；Ximing Lu；Jaehun Jung；Hyunwoo Kim；Amlan Kar；Sanja Fidler；Yejin Choi","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","trajectory_value"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","scaling_report"],"domains":["multimodal","reasoning"],"tags":["mcts","visual-reasoning","subquestions","test-time-search","agreement"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["A test-time MCTS that makes subquestion generation and agreement-weighted visual reasoning explicit.","Socratic-MCTS 在冻结的视觉语言模型输出中插入子问题—子答案对，以 MCTS 方式延展视觉推理，并用内部一致性投票选择答案。"],"why":"It broadens Track 5 beyond text math: visual search traces need node-level prompts, answers, agreement scores, and stopping data to be auditable.","primary_link":"https://aclanthology.org/2025.emnlp-main.1230/","links":[],"link_count":2,"sections":9},{"id":"soleval-solidity-smart-contract-generation-2025","title":"SolEval: Benchmarking Large Language Models for Repository-level Solidity Smart Contract Generation","year":2025,"venue":"EMNLP 2025","authors":["Zhiyuan Peng","Xin Yin","Rui Qian","Peiqin Lin","YongKang Liu","Hao Zhang","Chenhao Ying","Yuan Luo"],"authors_zh":"Zhiyuan Peng, Xin Yin, Rui Qian, Peiqin Lin, YongKang Liu, Hao Zhang, Chenhao Ying, Yuan Luo","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","smart-contracts","software-security"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"仓库级 Solidity 智能合约生成基准","best_for_zh":"需要以功能、安全与资源开销共同评测或训练智能合约代码模型的研究者。","confidence":"high","one_line":["SolEval releases 1,507 Solidity repository tasks whose generated contracts are jointly checked by Foundry tests, Slither scans, and gas profiling.","SolEval 从 28 个 Solidity 仓库发布 1,507 个任务，并以 Foundry 测试、Slither 安全扫描和 gas 指标联合验证合约生成。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://aclanthology.org/2025.emnlp-main.218/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pzy2000/SolEval"},{"key":"data","label":["Data","数据"],"url":"https://github.com/pzy2000/SolEval/tree/master/data"}],"link_count":3,"sections":9},{"id":"sota-with-less-2025","title":"SoTA with Less: MCTS-Guided Sample Selection for Data-Efficient Visual Reasoning Self-Improvement","year":2025,"venue":"NeurIPS 2025 Spotlight","authors":["Xiyao Wang","Zhengyuan Yang","Chao Feng","Hongjin Lu","Linjie Li","Chung-Ching Lin","Kevin Lin","Furong Huang","Lijuan Wang"],"authors_zh":"Xiyao Wang、Zhengyuan Yang、Chao Feng、Hongjin Lu、Linjie Li、Chung-Ching Lin、Kevin Lin、Furong Huang、Lijuan Wang","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","model_report","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["multimodal_reasoning","visual_mathematics","natural_image_understanding","chart_understanding","science_qa"],"tags":["mcts-data-selection","difficulty-filtering","multimodal-reasoning","visual-rlvr","open-data","model-specific-curriculum","self-improvement","grpo","release-audit"],"status":"partial","priority":"必读","paper_type_zh":"视觉推理 MCTS 难度筛选、RLVR 配方与开放数据审计","best_for_zh":"研究多模态 RLVR、策略相对难度、样本选择、数据谱系和 verifier 风险的读者","confidence":"high","one_line":["ThinkLite-VL uses model-specific MCTS solve depth and unsolved-after-50 status to select 11K/7.5K visual-reasoning prompts from a 70K pool for one-stage GRPO self-improvement.","ThinkLite-VL 以模型特定的 MCTS 求解深度与 50 轮未解状态，从 69,997 条视觉推理候选中选出 11K/7.5K 提示用于 GRPO；价值在难度感知筛选，边界是判别器与决策谱系未完整公开。"],"why":"It demonstrates a concrete search-budgeted alternative to accuracy-only filtering and releases the 70K pool plus the 7B-selected subset, while also showing why selection decisions, upstream licenses, failed traces, model revisions and rejected examples must be published for a difficulty-curation recipe to be reproducible and auditable.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/ac3cea0be817ebac21299b77fd114ddf-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/si0wang/ThinkLite-VL"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/russwang/ThinkLite-VL-70k"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/russwang/ThinkLite-VL-hard-11k"}],"link_count":8,"sections":9},{"id":"soundmind-2025","title":"SoundMind: RL-Incentivized Logic Reasoning for Audio-Language Models","year":2025,"venue":"EMNLP 2025 Main Conference","authors":["Xingjian Diao","Chunhui Zhang","Keyi Kong","Weiyi Wu","Chiyu Ma","Zhongyu Ouyang","Peijun Qing","Soroush Vosoughi","Jiang Gui"],"authors_zh":"Xingjian Diao、Chunhui Zhang、Keyi Kong、Weiyi Wu、Chiyu Ma、Zhongyu Ouyang、Peijun Qing、Soroush Vosoughi、Jiang Gui","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["audio-logical-reasoning","multimodal-chain-of-thought","natural-language-inference"],"tags":["audio-reasoning-data","multimodal-chain-of-thought","rule-reward","arxiv-2506.12935","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放音频推理数据集与强化学习研究","best_for_zh":"构建与审计语音思维链监督微调或规则奖励训练数据","confidence":"high","one_line":["SoundMind converts 6,446 logical triplets into aligned text and speech chain-of-thought demonstrations with entailment labels for audio-language SFT and rule-reward training.","SoundMind 将 6,446 个逻辑三元组转为文本与语音对齐的长推理示范和蕴含标签，用于音频语言模型的监督微调与规则奖励训练。"],"why":"Audio-language reasoning lacked an open corpus that aligns spoken questions, long spoken reasoning, text annotations, and objectively checkable final labels.","primary_link":"https://arxiv.org/abs/2506.12935","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xid32/SoundMind"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SoundMind-RL/SoundMindDataset"}],"link_count":5,"sections":9},{"id":"sparq-quality-diversity-2025","title":"SPARQ: Synthetic Problem Generation for Reasoning via Quality-Diversity Algorithms","year":2025,"venue":"arXiv","authors":["Alex Havrilla","Edward Hughes","Mikayel Samvelyan","Jacob Abernethy"],"authors_zh":"Alex Havrilla、Edward Hughes、Mikayel Samvelyan、Jacob Abernethy","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","scaling_report"],"domains":["mathematics","synthetic_reasoning"],"tags":["sparq","synthetic-problems","quality-diversity","solve-rate","self-improvement","math"],"status":"partial","priority":"可读","paper_type_zh":"质量—多样性合成数学问题构造与规模研究","best_for_zh":"研究合成题目、策略相对难度筛选、质量—多样性搜索和数学 SFT 数据构造的读者","confidence":"high","one_line":["SPARQ recursively mutates MATH problems, scores each child with 16 student rollouts, and turns successful student solutions from nontrivial children into SFT tuples organized by learned skill cells.","SPARQ 递归变异 MATH 题目，用 16 条学生 rollout 估计子题难度，并将非平凡子题的成功学生解答组织成按技能单元索引的 SFT 元组。"],"why":"It makes problem generation, difficulty filtering, diversity sampling, and tuple assembly explicit while exposing invalid-problem propagation, verifier dependence, yield, and missing-release risks.","primary_link":"https://arxiv.org/abs/2506.06499","links":[],"link_count":2,"sections":9},{"id":"sparsepo-token-masks-2025","title":"SparsePO: Controlling Preference Alignment of LLMs via Sparse Token Masks","year":2025,"venue":"Findings of EMNLP 2025","authors":["Fenia Christopoulou","Ronald Cardenas","Gerasimos Lampouras","Haitham Bou-Ammar","Jun Wang"],"authors_zh":"Fenia Christopoulou, Ronald Cardenas, Gerasimos Lampouras, Haitham Bou-Ammar, Jun Wang","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction-tuning"],"tags":["preference-optimization","token-mask","sparse-training","dpo","alignment"],"status":"verified","priority":"必读","paper_type_zh":"稀疏 token 级偏好优化研究","best_for_zh":"研究 token 级奖励和正则化选择如何重塑偏好训练信号、又不收集新标注的读者。","confidence":"high","one_line":["SparsePO learns sparse token masks that independently weight reward and KL contributions inside preference optimization.","SparsePO 学习稀疏 token 掩码，在偏好优化中分别加权奖励项和 KL 项的贡献。"],"why":"It makes each token's reward and KL contribution an explicit learnable part of the training record rather than an implicit consequence of a response-level loss.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1389/","links":[],"link_count":2,"sections":9},{"id":"spc-self-play-critic-2025","title":"SPC: Evolving Self-Play Critic via Adversarial Games for LLM Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Jiaqi Chen","Bang Zhang","Ruotian Ma","Peisong Wang","Xiaodan Liang","Zhaopeng Tu","Xiaolong Li","Kwan-Yee K. Wong"],"authors_zh":"Jiaqi Chen、Bang Zhang、Ruotian Ma、Peisong Wang、Xiaodan Liang、Zhaopeng Tu、Xiaolong Li、Kwan-Yee K. Wong","tracks":["data_construction_open_release_recipes"],"source_role":["verifier_reward","construction_recipe","process_supervision","data_release"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward","process_reward"],"training_use":["sft","reward_modeling","process_supervision","rlvr","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","self_play_anchor","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["math"],"tags":["self-play","adversarial-data","process-critic","step-verification","synthetic-errors","reinforcement-learning","open-release"],"status":"partial","priority":"可读","paper_type_zh":"对抗自博弈过程批评器与数据构造方法","best_for_zh":"研究过程监督、验证器奖励、自博弈数据生成或步骤级测试时搜索的读者","confidence":"medium","one_line":["SPC bootstraps a step critic and error generator, then creates solver-filtered adversarial traces and binary rewards over two self-play rounds.","SPC 先用 PRM800K 与教师模型初始化步骤批评器和错误生成器，再通过求解器影响筛选与两轮对抗自博弈生成批评轨迹和二元奖励。"],"why":"It turns model-generated corruptions and game outcomes into renewable verifier-training data, while exposing solver dependence, co-adaptation, and gated-release risks.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/cb7baa005c239c1c7c4098c2a9e00450-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/chen-judge/SPC/"},{"key":"data","label":["Data","数据"],"url":"https://connecthkuhk-my.sharepoint.com/%3Af%3A/g/personal/jadge_connect_hku_hk/EkB9OYBHr_tGmGeJ5xxTncgBXFnln9nPP4jmCKNcQSSDIQ?e=oF7b6g"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/judge/SPC-Critic-2"},{"key":"project","label":["Project","项目主页"],"url":"https://chen-judge.github.io/SPC/"}],"link_count":8,"sections":9},{"id":"spa-direct-preference-annotation-2025","title":"Spread Preference Annotation: Direct Preference Judgment for Efficient LLM Alignment","year":2025,"venue":"ICLR 2025","authors":["Dongyoung Kim","Kimin Lee","Jinwoo Shin","Jaehyung Kim"],"authors_zh":"Dongyoung Kim、Kimin Lee、Jinwoo Shin、Jaehyung Kim（韩国科学技术院、延世大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["frontier_pipeline"],"domains":["alignment","instruction-following"],"tags":["preference-annotation","synthetic-data","dpo","alignment"],"status":"verified","priority":"必读","paper_type_zh":"自生成偏好标注与迭代对齐研究","best_for_zh":"适合只有少量人工偏好种子、但需要扩展训练记录的读者。","confidence":"high","one_line":["SPA expands a small preference seed with direct, model-logit preference judgments and noise-aware training.","SPA 用模型 logits 直接标注新生成的回答对，并以噪声感知方式迭代训练。"],"why":"It makes preference-label construction and label-confidence treatment explicit parts of the training pipeline.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/file/342e5fc02b86dec9b24e41b22968e539-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kingdy2002/SPA"}],"link_count":4,"sections":9},{"id":"spurious-rewards-2025","title":"Spurious Rewards: Rethinking Training Signals in RLVR","year":2025,"venue":"ICML 2026","authors":["Rulin Shao","Shuyue Stella Li","Rui Xin","Scott Geng","Yiping Wang","Sewoong Oh","Simon Shaolei Du","Nathan Lambert","Sewon Min","Ranjay Krishna","Yulia Tsvetkov","Hannaneh Hajishirzi","Pang Wei Koh","Luke Zettlemoyer"],"authors_zh":"Rulin Shao, Shuyue Stella Li, Rui Xin, Scott Geng, Yiping Wang, Sewoong Oh, Simon Shaolei Du, Nathan Lambert, Sewon Min, Ranjay Krishna, Yulia Tsvetkov, Hannaneh Hajishirzi, Pang Wei Koh, Luke Zettlemoyer","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["reward_verifier_layer"],"domains":["math","rlvr"],"tags":["RLVR","GRPO","reward-signal-audit","clipping-bias","model-family-generalization"],"status":"verified","priority":"必读","paper_type_zh":"RLVR 奖励信号与优化机制审计论文","best_for_zh":"需要判断 RLVR 增益是否真由奖励信息带来、或准备复现数学推理强化学习的人。","confidence":"high","one_line":["Controlled RLVR audit showing that GRPO can raise Qwen math scores under random or incorrect rewards by amplifying pretrained behaviors, while the effect often fails outside Qwen.","受控审计显示：GRPO 可在随机或错误奖励下放大 Qwen 的既有推理行为并提高数学分数，但这一效应常无法迁移到其他模型家族。"],"why":"It makes random and format-only reward baselines, model-family transfer, and optimizer details necessary checks before interpreting an RLVR gain as new reasoning ability.","primary_link":"https://openreview.net/pdf/408777dd5148509b911d4870c5b94db8a37f37a5.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ruixin31/Spurious_Rewards"},{"key":"data","label":["Data","数据"],"url":"https://github.com/ruixin31/Spurious_Rewards/tree/main/code/data"}],"link_count":4,"sections":9},{"id":"sqale-real-schema-text-to-sql-2025","title":"SQaLe: A Large Text-to-SQL Corpus Grounded in Real Schemas","year":2025,"venue":"AI for Tabular Data Workshop at EurIPS 2025","authors":["Cornelius Wolff","Daniel Gomm","Madelon Hulsebos"],"authors_zh":"Cornelius Wolff, Daniel Gomm, Madelon Hulsebos","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["text-to-sql","databases","code-generation"],"tags":["text-to-sql","databases","execution-verification","2025"],"status":"verified","priority":"必读","paper_type_zh":"基于真实数据库模式的可执行 Text-to-SQL 数据集","best_for_zh":"需要大规模、可执行验证的 Text-to-SQL 训练或评测数据的研究者。","confidence":"high","one_line":["SQaLe scales text-to-SQL to 517,676 execution-validated triples grounded in 135,875 real-derived database schemas.","SQaLe 基于 135,875 个真实来源数据库模式构建 517,676 条经执行验证的文本到 SQL 三元组。"],"why":"It exposes SQL execution as a reproducible terminal check over unusually large and structurally realistic schemas.","primary_link":"https://arxiv.org/abs/2602.22223","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/trl-lab/SQaLe-Library"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/trl-lab/SQaLe-text-to-SQL-dataset"}],"link_count":5,"sections":9},{"id":"sra-mcts-2025","title":"SRA-MCTS: Self-driven Reasoning Augmentation with Monte Carlo Tree Search for Code Generation","year":2025,"venue":"IJCAI 2025","authors":["Bin Xu","Yiguan Lin","Yinghao Li","Yang Gao"],"authors_zh":"Bin Xu；Yiguan Lin；Yinghao Li；Yang Gao","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","search_substrate","trace_writing"],"domains":["code","reasoning"],"tags":["mcts","code-generation","sft","self-evaluation","leetcode"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时 rollout、选择器、反馈信号和计算预算的读者","confidence":"high","one_line":["An end-to-end MCTS-to-SFT code-reasoning recipe with official code and a model release.","SRA-MCTS 从中高难 LeetCode 题目出发，在树中生成思考步骤和代码，并把选中路径转换为 question-thinking-code 的 SFT 记录。"],"why":"It illustrates which search-tree fields must be released to audit an MCTS data recipe; the public final records are not a substitute for the missing branch history.","primary_link":"https://www.ijcai.org/proceedings/2025/965","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DIRECT-BIT/SRA-MCTS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/BinXD/SRA-MCTS-Llama-3.1-8B"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/BinXD/SRA-MCTS-Llama-3.1-8B"}],"link_count":6,"sections":9},{"id":"start-self-taught-reasoner-tools-2025","title":"START: Self-taught Reasoner with Tools","year":2025,"venue":"EMNLP 2025 (Main Conference)","authors":["Chengpeng Li","Mingfeng Xue","Zhenru Zhang","Jiaxi Yang","Beichen Zhang","Bowen Yu","Binyuan Hui","Junyang Lin","Xiang Wang","Dayiheng Liu"],"authors_zh":"Chengpeng Li、Mingfeng Xue、Zhenru Zhang、Jiaxi Yang、Beichen Zhang、Bowen Yu、Binyuan Hui、Junyang Lin、Xiang Wang、Dayiheng Liu","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","agent_environment","model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","self_play_anchor","reward_verifier_layer","release_audit"],"domains":["mathematical_reasoning","code_reasoning","tool_integrated_reasoning"],"tags":["start","hint-infer","hint-rft","self-training","rejection-fine-tuning","tool-integrated-reasoning","Python-interpreter","corrective-trajectories","long-chain-of-thought","missing-release"],"status":"partial","priority":"可读","paper_type_zh":"工具推理轨迹构造 recipe 与模型研究","best_for_zh":"研究 tool-integrated reasoning、自训练轨迹筛选、Python 环境反馈或审计未发布数据 recipe 的读者","confidence":"high","one_line":["START inserts diverse hints into long reasoning, filters Python-interleaved trajectories for corrective success, trains a 10K-seed intermediate model, and expands to a reported 40K self-training set, while releasing none of the records, code, environment, or checkpoints.","START 通过在长推理中插入多类 hint、调用 Python，并筛选由错转对的工具轨迹，先构造 10K D_seed 再扩展到报告的 40K D_START；其构造思路可供研究，但数据、checker、环境、模型和代码均未发布，且来源计数存在内部矛盾。"],"why":"It shows how tool feedback can become reusable supervised reasoning data without a separate teacher solver, but its unreleased verifier and environment, hidden rejected traces, licensing gaps, and inconsistent source totals make the recipe substantially less auditable than its aggregate results suggest.","primary_link":"https://aclanthology.org/2025.emnlp-main.683.pdf","links":[],"link_count":6,"sections":9},{"id":"steca-step-level-trajectory-calibration-2025","title":"STeCa: Step-level Trajectory Calibration for LLM Agent Learning","year":2025,"venue":"Findings of ACL 2025","authors":["Hanlin Wang","Jian Wang","Chak Tou Leong","Wenjie Li"],"authors_zh":"Hanlin Wang, Jian Wang, Chak Tou Leong, Wenjie Li","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent-reasoning","embodied-agents","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"基于环境回报的智能体步骤级轨迹校准方法与数据资源论文","best_for_zh":"需要把失败智能体轨迹转化为可训练步骤级纠错样本的研究者。","confidence":"high","one_line":["STeCa releases and constructs calibrated agent trajectories that attach deviation-aware corrections to environment-grounded steps.","通过定位偏离专家轨迹的动作并生成反思与校准轨迹，STeCa 将终局环境回报转化为可用于智能体学习的步骤级监督。"],"why":"It exposes a reusable route from terminal environment feedback to step-level trajectory supervision.","primary_link":"https://aclanthology.org/2025.findings-acl.604/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WangHanLinHenry/STeCa"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/uc?id=1tqZyqzyE7NnJdUJlwwfcwnPUFMafKdfA"}],"link_count":5,"sections":9},{"id":"stelar-vision-self-topology-aware-efficient-learning-for-aligned-reasoning-in-vision-2025","title":"STELAR-VISION: Self-Topology-Aware Efficient Learning for Aligned Reasoning in Vision","year":2025,"venue":"AAAI 2026","authors":["Chen Li","Han Zhang","Zhantao Yang","Fangyi Chen","Zihan Wang","Anudeepsekhar Bolimera","Marios Savvides"],"authors_zh":"Chen Li、Han Zhang、Zhantao Yang、Fangyi Chen、Zihan Wang、Anudeepsekhar Bolimera、Marios Savvides","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"视觉推理偏好数据与对齐训练论文","best_for_zh":"研究视觉语言模型、合成偏好数据或高效推理的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","STELAR-VISION 发布拓扑感知的视觉推理偏好数据，以兼顾多模态推理正确性与输出效率。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2508.08688","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Stellar-Neuron/STELAR-topo_vision_reasoning_preference_123k"}],"link_count":2,"sections":9},{"id":"hybrid-step-verifier-tts-2025","title":"Step-level Verifier-guided Hybrid Test-Time Scaling for Large Language Models","year":2025,"venue":"EMNLP 2025","authors":["Kaiyan Chang","Yonghao Shi","Chenglong Wang","Hang Zhou","Chi Hu","Xiaoqian Liu","Yingfeng Luo","Yuan Ge","Tong Xiao","JingBo Zhu"],"authors_zh":"Kaiyan Chang、Yonghao Shi、Chenglong Wang、Hang Zhou、Chi Hu、Xiaoqian Liu、Yingfeng Luo、Yuan Ge、Tong Xiao、JingBo Zhu（机构：东北大学、NiuTrans Research、字节跳动）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["reasoning"],"tags":["test-time-compute","process-verifier","self-refinement","hybrid-scaling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"免训练的测试时扩展研究（EMNLP 2025）","best_for_zh":"在固定部署预算下设计验证器引导推理的读者。","confidence":"high","one_line":["A process verifier decides where to refine reasoning steps and how to combine sequential and parallel inference-time scaling.","该方法用过程验证器决定何处细化推理步骤，并组合串行与并行的推理时扩展。"],"why":"It makes the allocation of inference computation conditional on step-level process evidence rather than only final-answer ranking.","primary_link":"https://aclanthology.org/2025.emnlp-main.931/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Lucky-259/Hybrid_TTS"}],"link_count":4,"sections":9},{"id":"stepsearch-2025","title":"StepSearch: Igniting LLMs Search Ability via Step-Wise Proximal Policy Optimization","year":2025,"venue":"EMNLP 2025","authors":["Xuhui Zheng","Kang An","Ziliang Wang","Yuhang Wang","Yichao Wu"],"authors_zh":"Xuhui Zheng, Kang An, Ziliang Wang, Yuhang Wang, Yichao Wu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["agent_training","test_time_compute"],"construction_layer":["search_substrate","scaling_report"],"domains":["reasoning"],"tags":["track5","raw_search_rollouts"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"medium","one_line":["StepSearch: Igniting LLMs Search Ability via Step-Wise Proximal Policy Optimization records subquestion search trajectories under information-gain and redundancy intermediate feedback.","StepSearch 以检索信息增益与冗余惩罚为每一步搜索提供 PPO 反馈，处理后的监督语料仍未核验公开。"],"why":"It makes step-wise PPO search action selection and its audit boundary visible for reasoning-data curation.","primary_link":"https://arxiv.org/abs/2505.15107","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zxh20001117/StepSearch"}],"link_count":3,"sections":9},{"id":"stepwise-reasoning-checkpoint-analysis-2025","title":"Stepwise Reasoning Checkpoint Analysis: A Test Time Scaling Method to Enhance LLMs' Reasoning","year":2025,"venue":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","authors":["Zezhong Wang","Xingshan Zeng","Weiwen Liu","Yufei Wang","Liangyou Li","Yasheng Wang","Lifeng Shang","Xin Jiang","Qun Liu","Kam-Fai Wong"],"authors_zh":"Zezhong Wang、Xingshan Zeng、Weiwen Liu、Yufei Wang、Liangyou Li、Yasheng Wang、Lifeng Shang、Xin Jiang、Qun Liu、Kam-Fai Wong","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","scaling_report"],"domains":["mathematics","reasoning","test_time_compute"],"tags":["stepwise-reasoning","checkpoint","answer-clustered-search","process-reward-model","test-time-scaling","candidate-augmentation","mathematical-reasoning"],"status":"partial","priority":"必读","paper_type_zh":"测试时扩展、过程奖励模型与推理搜索方法研究","best_for_zh":"研究 rollout 搜索、测试时计算、过程奖励模型、推理轨迹中间状态和可复现性审计的读者","confidence":"high","one_line":["SRCA probes a reasoning prefix with a temporary answer cue, uses checkpoint answers to diversify PRM-guided search and augment final candidates, but releases no code, traces, or replay ledger.","SRCA 在推理步骤间探测 checkpoint 答案，以其聚类维持 PRM 引导搜索的多样性并扩展最终候选；但没有公开代码、轨迹或可重放选择账本。"],"why":"It gives the rollout-search track an explicit intermediate-state object and selection contract, while showing that test-time accuracy cannot establish checkpoint-label quality, PRM calibration, or reproducibility.","primary_link":"https://aclanthology.org/2025.emnlp-main.866/","links":[],"link_count":4,"sections":9},{"id":"sticker-tts-historical-experience-2025","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","year":2025,"venue":"EMNLP 2025","authors":["Jie Chen","Jinhao Jiang","Yingqian Min","Zican Dong","Shijie Wang","Wayne Xin Zhao","Ji-Rong Wen"],"authors_zh":"Jie Chen、Jinhao Jiang、Yingqian Min、Zican Dong、Shijie Wang、Wayne Xin Zhao、Ji-Rong Wen（机构：中国人民大学高瓴人工智能学院、东北大学秦皇岛分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report","optimizer_scaffold"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","historical-experience","iterative-refinement","multi-agent","mathematical-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"测试时扩展与历史经验感知的迭代推理框架","best_for_zh":"研究固定推理预算下如何复用既有推理尝试的读者。","confidence":"high","one_line":["Sticker-TTS turns previous reasoning traces into corrected compact cues that guide later test-time solution attempts.","Sticker-TTS 将先前推理轨迹压缩为可校正的关键线索，用以引导后续测试时解题。"],"why":"It gives historical reasoning an explicit compact representation that can be corrected and reused instead of repeatedly sampling from scratch.","primary_link":"https://aclanthology.org/2025.emnlp-main.621/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCAIBox/Sticker-TTS"}],"link_count":3,"sections":9},{"id":"stop-overthinking-efficient-reasoning-survey-2025","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","year":2025,"venue":"Transactions on Machine Learning Research","authors":["Yang Sui","Yu-Neng Chuang","Guanchu Wang","Jiamu Zhang","Tianyi Zhang","Jiayi Yuan","Hongyi Liu","Andrew Wen","Shaochen Zhong","Na Zou","Hanjie Chen","Xia Hu"],"authors_zh":"Yang Sui 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["test_time_compute","evaluation"],"construction_layer":["trace_writing"],"domains":["reasoning","reasoning-data","chain-of-thought","inference-efficiency","evaluation"],"tags":["foundations-and-primers","efficient-reasoning","chain-of-thought","tmlr-2025","survey"],"status":"verified","priority":"必读","paper_type_zh":"高效推理综述","best_for_zh":"比较推理过程、预算与评测方法的读者。","confidence":"high","one_line":["A TMLR survey of how to reduce redundant reasoning without treating trace length as the only goal.","从模型、输出与提示三层整理高效推理的 TMLR 2025 综述。"],"why":"It links trace data, adaptive budgets, and evaluation rather than treating efficiency as mere token reduction.","primary_link":"https://arxiv.org/abs/2503.16419","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Eclipsess/Awesome-Efficient-Reasoning-LLMs"}],"link_count":4,"sections":9},{"id":"storm-born-2025","title":"STORM-BORN: A Challenging Mathematical Derivations Dataset Curated via a Human-in-the-Loop Multi-Agent Framework","year":2025,"venue":"Findings of ACL 2025","authors":["Wenhao Liu","Zhenyi Lu","Xinyu Hu","Jierui Zhang","Dailin Li","Jiacheng Cen","Huilin Cao","Haiteng Wang","Yuhan Li","Kun Xie","Dandan Li","Pei Zhang","Chengbo Zhang","Yuxiang Ren","Xiaohong Huang","Yan Ma"],"authors_zh":"Wenhao Liu、Zhenyi Lu、Xinyu Hu、Jierui Zhang、Dailin Li、Jiacheng Cen、Huilin Cao、Haiteng Wang、Yuhan Li、Kun Xie、Dandan Li、Pei Zhang、Chengbo Zhang、Yuxiang Ren、Xiaohong Huang、Yan Ma","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","benchmark","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["mathematics","mathematical_derivation","scientific_papers"],"tags":["storm-born","mathematical-derivations","multi-agent","human-in-the-loop","expert-curation"],"status":"partial","priority":"必读","paper_type_zh":"数学推导数据集、专家评测基准与人机协同构建配方","best_for_zh":"研究文档溯源推理数据、多智能体构造、专家判断、SFT 与科学数学评测审计的读者","confidence":"high","one_line":["A 100-row expert-selected derivation dataset built from a 2,000-pair multi-agent pool, released with source IDs, long answers, a 73/27 split, and four-option variants but without full lineage or the candidate pool.","STORM-BORN 从 2,000 条多智能体生成问答中由专家筛选出 100 条数学推导记录，并发布 73/27 切分与四选一变体；但完整谱系、候选池、人工协议与许可仍缺失。"],"why":"STORM-BORN makes full scientific documents and expert judgment central to reasoning-data construction, while the live source mismatch, shared-source split, missing review history, partial code, and unresolved licensing show the audit cost of document-derived supervision.","primary_link":"https://aclanthology.org/2025.findings-acl.1227/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lwhere/STORM-BORN"},{"key":"data","label":["Data","数据"],"url":"https://github.com/lwhere/STORM-BORN/tree/main/data"}],"link_count":6,"sections":9},{"id":"subliminal-learning-2025","title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data","year":2025,"venue":"Nature 2026; arXiv:2507.14805","authors":["Alex Cloud","Minh Le","James Chua","Jan Betley","Anna Sztyber-Betley","Jacob Hilton","Samuel Marks","Owain Evans"],"authors_zh":"Alex Cloud, Minh Le, James Chua, Jan Betley, Anna Sztyber-Betley, Jacob Hilton, Samuel Marks, Owain Evans","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["distillation","evaluation","audit"],"construction_layer":["release_audit","trace_writing"],"domains":["synthetic-data","lineage","distillation"],"tags":["synthetic-data","distillation","hidden-trait-transfer","lineage","filtering"],"status":"verified","priority":"必读","paper_type_zh":"审计、污染或验证器鲁棒性论文","best_for_zh":"想复盘验证器、基准污染、奖励投机或可复现性风险的研究者。","confidence":"high","one_line":["Subliminal Learning shows that teacher models can transmit behavioral traits through semantically unrelated generated data, even after visible trait references are filtered.","即使训练样本只含数字、代码或推理轨迹，且过滤了特征词，教师模型的偏好或失调仍可沿同基础模型蒸馏链传给学生。"],"why":"It is a data-lineage warning for reasoning distillation: synthetic traces may carry hidden model traits that are invisible to content filters.","primary_link":"https://arxiv.org/abs/2507.14805","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MinhxLe/subliminal-learning"},{"key":"project","label":["Project","项目主页"],"url":"https://subliminal-learning.com/"}],"link_count":4,"sections":9},{"id":"supergpqa-2025","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","year":2025,"venue":"arXiv preprint","authors":["M-A-P Team","Xinrun Du","Yifan Yao","Kaijing Ma","Bingli Wang","Tianyu Zheng","Kang Zhu","Minghao Liu","and 89 additional listed authors"],"authors_zh":"M-A-P Team 等（M-A-P、ByteDance Seed、2077.AI 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["specialized-academic-knowledge","domain-expert-benchmark"],"tags":["benchmark","domain_expert_benchmark","specialized-academic-knowledge"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["SuperGPQA exposes graduate-level QA across many disciplines as an auditable evaluation surface.","SuperGPQA 把横跨众多研究生学科的高难度问答做成可审计的评测面。"],"why":"Broad expert-domain coverage can complement GPQA after source/artifact audit.","primary_link":"https://arxiv.org/abs/2502.14739","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SuperGPQA/SuperGPQA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/m-a-p/SuperGPQA"},{"key":"project","label":["Project","项目主页"],"url":"https://supergpqa.github.io/"}],"link_count":5,"sections":9},{"id":"swe-bench-live-2025","title":"SWE-bench Goes Live!","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks / arXiv","authors":["Linghao Zhang","Shilin He","Chaoyun Zhang","Yu Kang","Bowen Li","Chengxing Xie","Junhao Wang","Maoquan Wang","Yufan Huang","Shengyu Fu","Elsie Nallipogu","Qingwei Lin","Yingnong Dang","Saravan Rajmohan","Dongmei Zhang"],"authors_zh":"Linghao Zhang 等（Microsoft）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces","programmatically_verifiable_outcome_data"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["swe-agents","live-github-issues","repository-repair"],"tags":["agent_environment","trajectory_data","swe-agents","live-github-issues","repository-repair"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2025 D&B / arXiv 的 live repository-repair benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["SWE-bench-Live refreshes repository repair evaluation with fresh GitHub issues.","SWE-bench-Live 用新鲜 GitHub issue 刷新仓库修复评测。"],"why":"fresh issue collection reduces contamination and tests environment construction","primary_link":"https://arxiv.org/abs/2505.23419","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/SWE-bench-Live"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-bench-Live/SWE-bench-Live"},{"key":"project","label":["Project","项目主页"],"url":"https://swe-bench-live.github.io/"}],"link_count":5,"sections":9},{"id":"swe-bench-pro-long-horizon-engineering-2025","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","year":2025,"venue":"ICLR 2026","authors":["Xiang Deng","Jeff Da","Edwin Pan","Yannis Yiming He","Charles Ide","Kanak Garg","Niklas Lauffer","Andrew Park","Nitin Pasari","Chetan Rane","Karmini Sampath","Maya Krishnan","Srivatsa Kundurthy","Sean Hendryx","Zifan Wang","Chen Bo Calvin Zhang","Noah Jacobson","Bing Liu","Brad Kenstler"],"authors_zh":"Xiang Deng, Jeff Da, Edwin Pan, Yannis Yiming He, Charles Ide, Kanak Garg, Niklas Lauffer, Andrew Park, Nitin Pasari, Chetan Rane, Karmini Sampath, Maya Krishnan, Srivatsa Kundurthy, Sean Hendryx, Zifan Wang, Chen Bo Calvin Zhang, Noah Jacobson, Bing Liu, Brad Kenstler","tracks":["programmatically_verifiable_outcome_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["software-engineering","long-horizon","docker","benchmark","2025"],"status":"verified","priority":"必读","paper_type_zh":"长程软件工程 benchmark 与可执行补丁评测数据","best_for_zh":"需要长程软件工程任务、F2P/P2P 测试或 benchmark 审计案例的研究者。","confidence":"high","one_line":["SWE-Bench Pro offers long-horizon engineering tasks with human-augmented specifications and Dockerized F2P/P2P tests, while its later audit exposes substantial task-quality risk.","SWE-Bench Pro 以人工补全的长程工程需求和 Docker 化 F2P/P2P 测试评测补丁，但后续官方审计指出其中相当一部分任务存在质量问题。"],"why":"It combines an executable professional-task benchmark with a concrete, consequential example of why verifier and task audits must remain part of reuse decisions.","primary_link":"https://arxiv.org/abs/2509.16941","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/scaleapi/SWE-bench_Pro-os"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScaleAI/SWE-bench_Pro"},{"key":"project","label":["Project","项目主页"],"url":"https://labs.scale.com/papers/swe-bench-pro"}],"link_count":6,"sections":9},{"id":"swe-bench-a-framework-for-the-scalable-generation-of-software-engineering-benchmarks-fro","title":"SWE-Bench++: A Framework for the Scalable Generation of Software Engineering Benchmarks from Open-Source Repositories","year":2025,"venue":"arXiv","authors":["Lilin Wang","Lucas Ramalho","Alan Celestino","Phuc Anthony Pham","Yu Liu","Umang Kumar Sinha","Andres Portillo","Onassis Osunwa","Gabriel Maduekwe"],"authors_zh":"Lilin Wang, Lucas Ramalho, Alan Celestino, Phuc Anthony Pham, Yu Liu, Umang Kumar Sinha, Andres Portillo, Onassis Osunwa, Gabriel Maduekwe","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** SWE-Bench++ automatically constructs 11,133 multilingual tasks, while its current downloadable public set contains 500.","SWE-Bench++ 自动构造 11,133 条多语言任务，但当前公开下载集为其中 500 条。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2512.17419","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TuringEnterprises/SWE-Bench-plus-plus"},{"key":"project","label":["Project","项目主页"],"url":"https://research.turing.com/swebench"}],"link_count":3,"sections":9},{"id":"swe-dev-2025","title":"SWE-Dev: Building Software Engineering Agents with Training and Inference Scaling","year":2025,"venue":"Findings of the Association for Computational Linguistics: ACL 2025","authors":["Haoran Wang","Zhenyu Hou","Yao Wei","Jie Tang","Yuxiao Dong"],"authors_zh":"Haoran Wang、Zhenyu Hou、Yao Wei、Jie Tang、Yuxiao Dong","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","construction_recipe","agent_environment","scaling_study"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["software_engineering","repository_level_code","agentic_reasoning"],"tags":["swe-dev","software-engineering-agents","executable-tests","fail-to-pass","agent-trajectories","openhands","trajectory-filtering","swe-bench-verified"],"status":"partial","priority":"必读","paper_type_zh":"软件工程智能体训练数据、可执行测试构造与扩展研究","best_for_zh":"研究仓库级代码智能体后训练、fail-to-pass 验证、轨迹筛选、训练/推理扩展以及数据可审计性的读者","confidence":"high","one_line":["SWE-Dev constructs repository repair tasks and fail-to-pass tests, releases 20,147 SFT/RFT-format OpenHands records, and studies trajectory and iteration scaling, while row-level lineage and replay remain incomplete.","SWE-Dev 从仓库修复任务构造 fail-to-pass 测试并发布 20,147 条 SFT/RFT 格式的 OpenHands 记录，用于研究训练与迭代扩展；但逐条来源、环境重放和污染审计仍不完整。"],"why":"It makes an executable-feedback pipeline for software-agent post-training inspectable, but also shows why released message traces, scores, and benchmark gains are insufficient without source, environment, and contamination evidence.","primary_link":"https://aclanthology.org/2025.findings-acl.193/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/SWE-Dev"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zai-org/SWE-Dev-train"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zai-org/SWE-Dev-32B"}],"link_count":7,"sections":9},{"id":"swe-dev-feature-driven-software-development-2025","title":"SWE-Dev: Evaluating and Training Autonomous Feature-Driven Software Development","year":2025,"venue":"arXiv","authors":["Yaxin Du","Yuzhu Cai","Yifan Zhou","Cheng Wang","Yu Qian","Xianghe Pang","Qian Liu","Yue Hu","Siheng Chen"],"authors_zh":"Yaxin Du, Yuzhu Cai, Yifan Zhou, Cheng Wang, Yu Qian, Xianghe Pang, Qian Liu, Yue Hu, Siheng Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["software-engineering","unit-tests","feature-development","agent-training","2025"],"status":"verified","priority":"可读","paper_type_zh":"可执行功能开发数据集、软件工程 benchmark 与训练环境","best_for_zh":"需要仓库级功能开发任务、单测奖励或软件工程 agent 训练数据的研究者。","confidence":"high","one_line":["SWE-Dev supplies 14,000 runnable feature-development training tasks and 500 test tasks where developer-authored unit tests score repository-level implementations.","SWE-Dev 提供 1.4 万个可运行的功能开发训练任务和 500 个测试任务，并以开发者编写的单测验证仓库级实现。"],"why":"Its runnable test suites turn a repository-level development outcome into an executable feedback signal rather than a text-only code comparison.","primary_link":"https://arxiv.org/abs/2505.16975","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DorothyDUUU/SWE-Dev"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Dorothydu/SWE-Dev/tree/main/dataset"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/Dorothydu/SWE-Dev"}],"link_count":6,"sections":9},{"id":"swe-lancer-freelance-software-engineering-2025","title":"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?","year":2025,"venue":"ICML 2025","authors":["Samuel Miserendino","Michele Wang","Tejal Patwardhan","Johannes Heidecke"],"authors_zh":"Samuel Miserendino, Michele Wang, Tejal Patwardhan, Johannes Heidecke","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","agent_training"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","agent-evaluation"],"tags":["software-engineering","freelance-work","docker","agent-evaluation","2025"],"status":"verified","priority":"必读","paper_type_zh":"真实自由职业软件工程任务的可执行与决策评测基准","best_for_zh":"需要把代码执行结果和真实工程决策联系起来评测智能体的研究者。","confidence":"high","one_line":["SWE-Lancer evaluates agents on more than 1,400 real freelance engineering tasks using triple-verified tests or the original manager's hiring choice.","SWE-Lancer 以 1,400 余项真实自由职业软件任务评测智能体，独立工程任务由三重核验测试判分，管理任务对照原始经理选择。"],"why":"It evaluates practical engineering work through executable task contracts or the decisions that originally allocated the work.","primary_link":"https://arxiv.org/abs/2502.12115","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/preparedness"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/swe-lancer/"}],"link_count":5,"sections":9},{"id":"swe-mera-2025","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","year":2025,"venue":"EMNLP 2025 System Demonstrations","authors":["Pavel Adamenko","Mikhail Ivanov","Aidar Valeev","Rodion Levichev","Pavel Zadorozhny","Ivan Lopatin","Dmitrii Babaev","Alena Fenogenova","Valentin Malykh"],"authors_zh":"Pavel Adamenko、Mikhail Ivanov、Aidar Valeev、Rodion Levichev、Pavel Zadorozhny、Ivan Lopatin、Dmitrii Babaev、Alena Fenogenova、Valentin Malykh","tracks":["environment_agent_trajectory_data","programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software_engineering","repository_repair","github_issues"],"tags":["environment-agent-trajectory-data","agent-trajectories"],"status":"partial","priority":"必读","paper_type_zh":"动态仓库修复基准、执行环境与终端验证审计","best_for_zh":"研究 SWE agents、动态基准、Docker 重放、测试验证和污染控制的读者","confidence":"medium","one_line":["SWE-MERA releases refreshable repository-repair records with commits, patches, tests, commands, and environment metadata, but its pinned public checker marks solved using PASS_TO_PASS only and does not enforce FAIL_TO_PASS.","SWE-MERA 持续发布带 base commit、补丁、测试和环境命令的仓库修复任务，但固定版本的公开 checker 仅检查 PASS_TO_PASS，没有强制 FAIL_TO_PASS。"],"why":"It shows why dynamic task freshness, dataset versioning, container replay, and exact terminal semantics must be audited independently of a benchmark's stated design.","primary_link":"https://aclanthology.org/2025.emnlp-demos.30/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MERA-Evaluation/repotest"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MERA-evaluation/SWE-MERA"},{"key":"project","label":["Project","项目主页"],"url":"https://mera-evaluation.github.io/demo-swe-mera/"}],"link_count":8,"sections":9},{"id":"swe-mirror-2025","title":"SWE-Mirror: Scaling Issue-Resolving Datasets by Mirroring Issues Across Repositories","year":2025,"venue":"arXiv preprint","authors":["Junhao Wang","Daoguang Zan","Shulin Xin","Siyao Liu","Yurong Wu","Kai Shen"],"authors_zh":"Junhao Wang、Daoguang Zan、Shulin Xin、Siyao Liu、Yurong Wu、Kai Shen","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","data_release"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent_trajectories","environment_interaction","software_engineering"],"tags":["environment-agent-trajectory-data","software-engineering","agent-trajectories","executable-tests","issue-mirroring"],"status":"needs_url","priority":"可读","paper_type_zh":"软件工程环境与智能体轨迹构造方法","best_for_zh":"研究可执行 SWE 任务生成、OpenHands 轨迹 SFT、测试 verifier 与数据污染审计的读者","confidence":"medium","one_line":["SWE-Mirror constructs 60,671 executable repair tasks by re-instantiating real GitHub issue semantics in 40 reusable repository Gyms, then reports an SFT mixture of 12,456 successful OpenHands trajectories.","SWE-Mirror 将真实 GitHub issue 的语义重建到 40 个可复用仓库 Gym 中，形成 60,671 个可执行修复任务和 12,456 条成功轨迹的 SFT 混合，但官方数据与代码发布入口尚未核实。"],"why":"It separates the costly repository Gym from the issue semantics being mirrored, providing a scalable programmatic feedback surface for software-engineering agent training while exposing clear semantic-fidelity, contamination, release, and replay risks.","primary_link":"https://arxiv.org/abs/2509.08724","links":[],"link_count":2,"sections":9},{"id":"swe-polybench-multilanguage-coding-agents-2025","title":"SWE-PolyBench: A multi-language benchmark for repository level evaluation of coding agents","year":2025,"venue":"arXiv","authors":["Muhammad Shihab Rashid","Christian Bock","Yuan Zhuang","Alexander Buchholz","Tim Esler","Simon Valentin","Luca Franceschi","Martin Wistuba","Prabhu Teja Sivaprasad","Woo Jung Kim","Anoop Deoras","Giovanni Zappella","Laurent Callot"],"authors_zh":"Muhammad Shihab Rashid, Christian Bock, Yuan Zhuang, Alexander Buchholz, Tim Esler, Simon Valentin, Luca Franceschi, Martin Wistuba, Prabhu Teja Sivaprasad, Woo Jung Kim, Anoop Deoras, Giovanni Zappella, Laurent Callot","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["evaluation","agent_training","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","agent-evaluation"],"tags":["software-engineering","multilingual","unit-tests","docker","2025"],"status":"verified","priority":"必读","paper_type_zh":"多语言仓库级代码 agent benchmark 与可执行补丁评测数据","best_for_zh":"需要多语言仓库修复评测、F2P/P2P 测试或 Docker 化复现实验的研究者。","confidence":"high","one_line":["SWE-PolyBench pairs 2,110 multi-language repository tasks with Dockerized F2P/P2P tests to measure whether coding-agent patches really resolve an issue.","SWE-PolyBench 为 2,110 个多语言仓库任务提供 Docker 环境与 F2P/P2P 测试，用可执行结果判定代码 agent 补丁是否真正修复问题。"],"why":"Its paired pass/fail test partitions and frozen environment make patch correctness auditable across different language ecosystems.","primary_link":"https://arxiv.org/abs/2504.08703","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/amazon-science/SWE-PolyBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AmazonScience/SWE-PolyBench"},{"key":"project","label":["Project","项目主页"],"url":"https://amazon-science.github.io/SWE-PolyBench/"}],"link_count":6,"sections":9},{"id":"swe-rebench-2025","title":"SWE-rebench: An Automated Pipeline for Task Collection and Decontaminated Evaluation of Software Engineering Agents","year":2025,"venue":"NeurIPS 2025 / arXiv","authors":["Ibragim Badertdinov","Alexander Golubev","Maksim Nekrashevich","Anton Shevtsov","Simon Karasik","Andrei Andriushchenko","Maria Trofimova","Daria Litvintseva","Boris Yangel"],"authors_zh":"Ibragim Badertdinov 等（Nebius）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["swe-agents","decontaminated-evaluation","task-mining"],"tags":["agent_environment","trajectory_data","swe-agents","decontaminated-evaluation","task-mining"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2025 / arXiv 的 SWE agent evaluation benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["SWE-rebench automates fresh SWE task collection and decontaminated agent evaluation.","SWE-rebench 自动收集新鲜 SWE 任务并进行去污染智能体评测。"],"why":"it automates contamination-aware SWE task mining and environment setup","primary_link":"https://arxiv.org/abs/2505.20411","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-rebench/SWE-bench-fork"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nebius/SWE-rebench"},{"key":"project","label":["Project","项目主页"],"url":"https://swe-rebench.com/"}],"link_count":5,"sections":9},{"id":"swe-sharp-bench-csharp-software-engineering-2025","title":"SWE-Sharp-Bench: A Reproducible Benchmark for C# Software Engineering Tasks","year":2025,"venue":"IEEE/ACM AIware 2025","authors":["Sanket Mhatre","Yasharth Bajpai","Sumit Gulwani","Emerson Murphy-Hill","Gustavo Soares"],"authors_zh":"Sanket Mhatre, Yasharth Bajpai, Sumit Gulwani, Emerson Murphy-Hill, Gustavo Soares","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","agent_training","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software-engineering","code-generation","performance-engineering"],"tags":["csharp","dotnet","software-engineering","unit-tests","2025"],"status":"verified","priority":"可读","paper_type_zh":"可程序验证的仓库级性能优化基准","best_for_zh":"需要以正确性测试和可复现运行时共同评测代码智能体的研究者。","confidence":"high","one_line":["SWE-Sharp-Bench brings reproducible, test-verified software-engineering tasks to 150 C#/.NET instances across 17 repositories.","SWE-Sharp-Bench 提供 150 个 C#/.NET 真实仓库任务，并以 .NET 测试执行验证智能体修复。"],"why":"It evaluates a patch through executable behavior and measured runtime rather than a text-only judgment.","primary_link":"https://arxiv.org/abs/2511.02352","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/microsoft/prose/tree/main/misc/SWE-Sharp-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/microsoft/SWE-Sharp-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://aka.ms/swesharparxiv"}],"link_count":5,"sections":9},{"id":"swe-smith-scaling-data-construction-for-software-engineering-agents-2025","title":"SWE-smith: Scaling Data for Software Engineering Agents","year":2025,"venue":"NeurIPS 2025","authors":["John Yang","Kilian Lieret","Carlos E. Jimenez","Alexander Wettig","Kabir Khandpur","Yanzhe Zhang","Binyuan Hui","Ofir Press","Ludwig Schmidt","Diyi Yang"],"authors_zh":"John Yang, Kilian Lieret, Carlos E. Jimenez, Alexander Wettig, Kabir Khandpur, Yanzhe Zhang, Binyuan Hui, Ofir Press, Ludwig Schmidt, Diyi Yang","tracks":["data_construction_open_release_recipes","training_usage_optimization_objectives","programmatically_verifiable_outcome_data","environment_agent_trajectory_data","benchmarks_evaluation_surfaces","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","agent_environment","scaling_study"],"verification_contract":["programmatic","environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["agent_training","sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","optimizer_scaffold","release_audit"],"domains":["software_engineering","code","python","agents"],"tags":["curated-card","primary-link-checked","artifact-verified","swe","agents","executable-tests","synthetic-data","open-release"],"status":"verified","priority":"必读","paper_type_zh":"软件工程环境与智能体训练数据构造管线","best_for_zh":"构建可执行 SWE-agent 任务、轨迹与训练环境的研究者","confidence":"high","one_line":["SWE-smith builds 50,137 executable Python repair tasks from test-breaking synthetic diffs and distills 5,016 successful SWE-agent episodes selected from 17,906 attempts.","SWE-smith 把 Python 仓库变成 Docker 化执行环境，并合成 5 万多条破坏测试的修复任务、轨迹和一个训练后的 SWE 智能体。"],"why":"It makes the environment, task-construction contract, executable feedback, and trajectory-selection boundary inspectable for the data-construction and open-release track.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/8b86cf5ace600c48fd188efbb8dedec8-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-bench/SWE-smith"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-bench/SWE-smith"},{"key":"project","label":["Project","项目主页"],"url":"https://swesmith.com/"}],"link_count":7,"sections":9},{"id":"swe-synth-verifiable-bug-fix-data-2025","title":"SWE-Synth: Synthesizing Verifiable Bug-Fix Data to Enable Large Language Models in Resolving Real-World Bugs","year":2025,"venue":"ICSE 2026 Research Track (Distinguished Paper Award)","authors":["Minh V. T. Pham","Huy N. Phan","Hoang Nhat Phan","Cuong Chi Le","Tien N. Nguyen","Nghi D. Q. Bui"],"authors_zh":"Minh V. T. Pham, Huy N. Phan, Hoang Nhat Phan, Cuong Chi Le, Tien N. Nguyen, Nghi D. Q. Bui","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","agent_training","evaluation"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["software-engineering","program-repair"],"tags":["program-repair","unit-tests","synthetic-data","repair-trajectories","2025"],"status":"verified","priority":"可读","paper_type_zh":"带可执行测试与修复轨迹的合成仓库级缺陷修复数据集","best_for_zh":"需要缺陷修复训练数据、可执行单测验证或修复轨迹的研究者。","confidence":"high","one_line":["SWE-Synth turns repository components into test-verifiable synthetic bugs and retains successful agent repair trajectories as bug-fix training data.","SWE-Synth 把仓库组件改写为可测试的合成缺陷，并保留通过测试的 agent 修复轨迹，形成可验证的缺陷修复训练数据。"],"why":"It preserves test-validated repair outcomes and intermediate repair actions, not just static before/after diffs.","primary_link":"https://arxiv.org/abs/2504.14757","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/swesynth/SWE-Synth"},{"key":"project","label":["Project","项目主页"],"url":"https://swesynth.com/"}],"link_count":5,"sections":9},{"id":"synlogic-2025","title":"SynLogic: Synthesizing Verifiable Reasoning Data at Scale for Learning Logical Reasoning and Beyond","year":2025,"venue":"NeurIPS 2025","authors":["Junteng Liu","Yuanxiang Fan","Zhuo Jiang","Han Ding","Yongyi Hu","Chi Zhang","Yiqi Shi","Shitong Weng","Aili Chen","Shiqi Chen","Yunan Huang","Mozhi Zhang","Pengyu Zhao","Junjie Yan","Junxian He"],"authors_zh":"Junteng Liu, Yuanxiang Fan, Zhuo Jiang, Han Ding, Yongyi Hu, Chi Zhang, Yiqi Shi, Shitong Weng, Aili Chen, Shiqi Chen, Yunan Huang, Mozhi Zhang, Pengyu Zhao, Junjie Yan, Junxian He","tracks":["programmatically_verifiable_outcome_data","data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["rlvr","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["logical-reasoning","synthetic-puzzles","rlvr"],"tags":["logical-reasoning","synthetic-data","rule-based-verifier","rlvr","open-release","difficulty-control","release-versioning"],"status":"partial","priority":"必读","paper_type_zh":"合成逻辑推理数据发布与 RLVR 构造配方","best_for_zh":"构造可控非数学 RLVR 任务和审计规则奖励的研究者","confidence":"medium","one_line":["SynLogic releases verifier-ready prompts and generator/verifier code for 35 logical tasks, with a 27-task Easy configuration and a 35-task Hard configuration, while online RL rollouts remain unreleased.","SynLogic 开放 35 类逻辑任务的可配置生成器、任务专用验证器和 Easy/Hard RLVR 提示数据，而其修订历史也说明可执行数据仍需版本化审计。"],"why":"It connects task sourcing, rule-based synthesis, model-calibrated difficulty, executable binary rewards, and DAPO-style RL beyond mathematics and code, while its release history illustrates why configuration identity, verifier behavior, and immutable revisions must be audited separately from benchmark gains.","primary_link":"https://openreview.net/forum?id=XtNiw8OQsy","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MiniMax-AI/SynLogic"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MiniMaxAI/SynLogic"}],"link_count":5,"sections":9},{"id":"swe-flow-test-driven-software-engineering-data-2025","title":"Synthesizing Software Engineering Data in a Test-Driven Manner","year":2025,"venue":"ICML 2025","authors":["Lei Zhang","Jiaxi Yang","Min Yang","Jian Yang","Mouxiang Chen","Jiajun Zhang","Zeyu Cui","Binyuan Hui","Junyang Lin"],"authors_zh":"Lei Zhang, Jiaxi Yang, Min Yang, Jian Yang, Mouxiang Chen, Jiajun Zhang, Zeyu Cui, Binyuan Hui, Junyang Lin","tracks":["programmatically_verifiable_outcome_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code","software-engineering"],"tags":["software-engineering","tdd","runtime-dependency-graph","execution","icml-2025"],"status":"verified","priority":"可读","paper_type_zh":"带阶段测试验证的测试驱动软件工程数据","best_for_zh":"研究过程监督、TDD 代理和软件工程执行奖励的研究者。","confidence":"high","one_line":["SWE-Flow turns unit tests and runtime dependency graphs into executable, incremental software-development tasks.","SWE-Flow 将单元测试和运行时依赖图转化为可执行、增量式的软件开发任务。"],"why":"It creates process-oriented software-engineering supervision where every stage has a concrete test verifier.","primary_link":"https://proceedings.mlr.press/v267/zhang25cn.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Hambaobao/SWE-Flow"}],"link_count":4,"sections":9},{"id":"swirl-2025","title":"Synthetic Data Generation & Multi-Step RL for Reasoning & Tool Use","year":2025,"venue":"COLM 2025","authors":["Anna Goldie","Azalia Mirhoseini","Hao Zhou","Irene Cai","Christopher D. Manning"],"authors_zh":"Anna Goldie, Azalia Mirhoseini, Hao Zhou, Irene Cai, Christopher D. Manning","tracks":["environment_agent_trajectory_data","training_usage_optimization_objectives"],"source_role":["construction_recipe","process_supervision","verifier_reward","agent_environment","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","state_action_level","full_episode","scalar_reward","process_reward"],"training_use":["sft","process_supervision","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report"],"domains":["multi_hop_question_answering","mathematical_reasoning","tool_use","search","synthetic_data","reinforcement_learning"],"tags":["environment-agent-trajectory-data","training-usage-optimization-objectives","swirl","synthetic-trajectories","step-wise-rl","process-filtering","generative-reward-model","tool-use","offline-rl"],"status":"partial","priority":"可读","paper_type_zh":"工具使用轨迹构造与离线逐步强化学习方法","best_for_zh":"研究智能体轨迹、过程筛选、逐步奖励和离线 RL 的读者","confidence":"high","one_line":["SWiRL trains Gemma 2 from fixed search/calculator trajectories by turning every action and its full prefix into a reward-bearing RL sample; the recipe and prompts are public, but the trajectories, reward logs, implementation, and checkpoints are not.","SWiRL 将离线搜索/计算器完整轨迹拆为动作结束的前缀子轨迹，以 Gemini 1.5 Pro 过程筛选和逐动作生成式奖励训练 Gemma 2；方法与提示公开，但轨迹、奖励日志、代码和模型未发布。"],"why":"It makes the supervision boundary explicit: a full tool-use episode is the source object, yet optimization consumes overlapping action-prefix subtrajectories with model-judged scalar feedback. That distinction is essential for auditing sample weighting, process-label quality, incorrect-outcome retention, and replayability.","primary_link":"https://openreview.net/forum?id=oN9STRYQVa","links":[],"link_count":3,"sections":9},{"id":"anthropic-claude-opus-4-1-system-card-2025","title":"System Card Addendum: Claude Opus 4.1","year":2025,"venue":"Anthropic system-card addendum","authors":["Anthropic"],"authors_zh":"Anthropic","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["agent_training","evaluation","audit","safety_alignment","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","coding","computer_use","tool_use","agentic_systems","safety","cybersecurity","biosecurity","alignment_audit","reward_hacking"],"tags":["anthropic","claude-opus-4-1","system-card-addendum","incremental-model-report","automated-behavioral-audit","model-generated-evaluation","model-as-judge","safety-prompts","synthetic-prompts","agentic-coding","computer-use","prompt-injection","reinforcement-learning","reward-hacking","training-distribution-evaluation","evaluation-awareness","corrected-metrics","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型增量系统卡与训练审计台账","best_for_zh":"研究模型生成审计、agentic safety、prompt injection、reward hacking 与增量模型披露的读者","confidence":"high","one_line":["The Opus 4.1 addendum adds a concrete 290-seed/1,160-transcript model-audit pipeline, specialized prompt-injection RL, agentic safety environments and training-distribution reward-hacking rates, while leaving the underlying data, scorers, rewards and model delta unreleased.","Claude Opus 4.1 增量系统卡披露 human/synthetic safety prompts、290 seeds 扩展为每模型 1,160 条 24–64-turn 审计轨迹、专门 prompt-injection RL 与 agentic 环境，但基础数据 delta、生成/评分偏差、reward 和记录均未开放。"],"why":"It shows how a frontier-model update can disclose useful evaluation and training-audit objects without becoming reproducible: the same Opus 4 family supplies the auditor and judges, extreme scenarios drive absolute scores, evaluation awareness threatens validity, and even training-environment hack rates lack the records and reward contracts needed for independent audit.","primary_link":"https://www-cdn.anthropic.com/9fa30625273bafdf5af82c93719d7ca606485a16.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://www.anthropic.com/news/claude-opus-4-1"}],"link_count":3,"sections":9},{"id":"t1-advancing-language-model-reasoning-2025","title":"T1: Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","year":2025,"venue":"Proceedings of the 42nd International Conference on Machine Learning (ICML 2025), PMLR 267","authors":["Zhenyu Hou","Xin Lv","Rui Lu","Jiajie Zhang","Yujiang Li","Zijun Yao","Juanzi Li","Jie Tang","Yuxiao Dong"],"authors_zh":"Zhenyu Hou、Xin Lv、Rui Lu、Jiajie Zhang、Yujiang Li、Zijun Yao、Juanzi Li、Jie Tang、Yuxiao Dong","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","rlvr","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["mathematics","reasoning","test_time_compute"],"tags":["t1","synthesized-cot","self-verification","trial-and-error","rlvr","rloo","high-temperature-rollouts","inference-scaling"],"status":"partial","priority":"可读","paper_type_zh":"数学推理数据构造、答案核验强化学习与推理时扩展研究","best_for_zh":"需要设计 RLVR、长推理轨迹、rollout 多样性或推理预算实验，并希望审计公开训练数据边界的研究者","confidence":"high","one_line":["T1 synthesizes trial-and-error/self-verification math chains for SFT and applies answer-verified, high-temperature 64-rollout RLOO to study training and inference scaling, but public release semantics, lineage, and replay remain incomplete.","T1 将公开数学题构造成含试错与自我核查的推理链，用答案核验的 64-rollout RLOO 研究训练和推理时扩展；但公开数据的语义、谱系与可复现性仍不完整。"],"why":"It is a concrete evidence source for coupling long reasoning traces with outcome verification and diversity-oriented RL; it is not yet a self-contained, auditable training dataset because the public files lack documented lineage, schema semantics, and verifier decisions.","primary_link":"https://proceedings.mlr.press/v267/hou25e.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/T1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zai-org/T1"}],"link_count":5,"sections":9},{"id":"t2-adaptive-tts-cqa-2025","title":"T2: An Adaptive Test-Time Scaling Strategy for Contextual Question Answering","year":2025,"venue":"EMNLP 2025","authors":["Zhengyi Zhao","Shubo Zhang","Zezhong Wang","Huimin Wang","Yutian Zhao","Bin Liang","Yefeng Zheng","Binyang Li","Kam-Fai Wong","Xian Wu"],"authors_zh":"Zhengyi Zhao、Shubo Zhang、Zezhong Wang、Huimin Wang、Yutian Zhao、Bin Liang、Yefeng Zheng、Binyang Li、Kam-Fai Wong、Xian Wu（机构：香港中文大学、国际关系学院、腾讯贾维斯实验室、西湖大学）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["question-answering","multi_hop_reasoning"],"tags":["test-time-compute","adaptive-reasoning","contextual-question-answering","strategy-selection","efficiency"],"status":"verified","priority":"可读","paper_type_zh":"面向上下文问答的自适应测试时扩展框架","best_for_zh":"研究检索支撑推理中的自适应预算与策略选择的读者。","confidence":"high","one_line":["T2 selects a question-specific reasoning strategy from analogous examples to spend more inference effort only where it is useful.","T2 从相似示例中选择针对问题的推理策略，只在确有必要时投入更多推理计算。"],"why":"It replaces a fixed think-long-or-think-short policy with per-question strategy selection and reports both quality and time.","primary_link":"https://aclanthology.org/2025.emnlp-main.185/","links":[],"link_count":2,"sections":9},{"id":"curriculum-distillation-small-models-2025","title":"Teach Small Models to Reason by Curriculum Distillation","year":2025,"venue":"EMNLP","authors":["Wangyi Jiang","Yaojie Lu","Hongyu Lin","Xianpei Han","Le Sun"],"authors_zh":"Wangyi Jiang、Yaojie Lu、Hongyu Lin、Xianpei Han、Le Sun","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics"],"tags":["curriculum-distillation","teacher-traces","difficulty-curriculum","long-cot","math-verify","multi-mode-generation"],"status":"partial","priority":"可读","paper_type_zh":"难度感知的课程蒸馏数据构建方案","best_for_zh":"研究小模型推理蒸馏、教师轨迹筛选与课程数据审计的读者","confidence":"high","one_line":["Curriculum Distillation filters a 6,445-problem matched teacher pool with Math-Verify and changes trace style across two difficulty-aware SFT stages, but releases neither the pool nor its construction artifacts.","该工作以 Math-Verify 筛得 6,445 道题的匹配教师池，并按难度在两阶段 SFT 中切换轨迹类型，但未发布语料与构建产物。"],"why":"It turns teacher mode, answer verification, difficulty, and training order into explicit data-construction variables; the unreleased examples, indices, prompts, and lineage keep it at recipe/audit-reference status rather than a reusable data release.","primary_link":"https://aclanthology.org/2025.emnlp-main.376/","links":[],"link_count":3,"sections":9},{"id":"target-dpo-focal-preference-alignment-code-2025","title":"Teaching Your Models to Understand Code via Focal Preference Alignment","year":2025,"venue":"EMNLP 2025","authors":["Jie Wu","Haoling Li","Xin Zhang","Xiao Liu","Yangyu Huang","Jianwen Luo","Yizhen Zhang","Zuchao Li","Ruihang Chu","Yujiu Yang","Scarlett Li"],"authors_zh":"Jie Wu、Haoling Li、Xin Zhang、Xiao Liu、Yangyu Huang、Jianwen Luo、Yizhen Zhang、Zuchao Li、Ruihang Chu、Yujiu Yang、Scarlett Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["pairwise_preference","step_level"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-engineering"],"tags":["code","dpo","iterative-debugging","preference-data","emnlp-2025"],"status":"verified","priority":"可读","paper_type_zh":"代码偏好数据集与细粒度 DPO 方法","best_for_zh":"需要依据测试反馈训练代码纠错模型、代码 DPO 或奖励模型的研究者","confidence":"high","one_line":["Target-DPO releases CodeFlow, 59,000 iterative-debugging code preference pairs that focus DPO on the tokens resolving test-verified errors.","CodeFlow 发布约 5.9 万代码偏好对，聚焦模型最易混淆的局部质量差异，适合代码推理 DPO 与奖励建模。"],"why":"It turns an iterative code repair trajectory into a preference record that isolates the error-resolving part of the edit.","primary_link":"https://aclanthology.org/2025.emnlp-main.707/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JieWu02/Target-DPO"}],"link_count":3,"sections":9},{"id":"testgeneval-unit-test-generation-2025","title":"TestGenEval: A Real World Unit Test Generation and Test Completion Benchmark","year":2025,"venue":"ICLR 2025","authors":["Kush Jain","Gabriel Synnaeve","Baptiste Rozière"],"authors_zh":"Kush Jain, Gabriel Synnaeve, Baptiste Rozière","tracks":["programmatically_verifiable_outcome_data","benchmarks_evaluation_surfaces"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-testing","unit-tests"],"tags":["programmatic-verification","benchmark","2025"],"status":"verified","priority":"可读","paper_type_zh":"真实仓库单元测试生成与补全基准","best_for_zh":"需要评测测试代码生成、覆盖率提升或变异测试表现的研究者。","confidence":"high","one_line":["TestGenEval evaluates unit-test generation and completion on 1,210 real repository file pairs using execution, coverage, and mutation-score feedback.","TestGenEval 从真实 Python 仓库构造 1,210 个代码—测试文件对，用执行通过率、覆盖率和 mutation score 评估测试生成与补全。"],"why":"It exposes a rerunnable outcome-verification surface rather than a text-only reference answer.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/26ded5c8ee8ec1bc4caced4e1c9b1584-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/testgeneval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/kjain14/testgeneval"},{"key":"project","label":["Project","项目主页"],"url":"https://testgeneval.github.io/"}],"link_count":6,"sections":9},{"id":"text2sql-flow-a-robust-sql-aware-data-augmentation-framework-for-text-to-sql","title":"Text2SQL-Flow: A Robust SQL-Aware Data Augmentation Framework for Text-to-SQL","year":2025,"venue":"arXiv","authors":["Qifeng Cai","Hao Liang","Chang Xu","Tao Xie","Wentao Zhang","Bin Cui"],"authors_zh":"Qifeng Cai, Hao Liang, Chang Xu, Tao Xie, Wentao Zhang, Bin Cui","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** An 89,544-example Text-to-SQL dataset built by six-dimensional SQL-aware augmentation and database execution.","用六维 SQL-aware 扩增和数据库执行验证构建 89,544 条 Text-to-SQL 数据。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2511.10192","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TechNomad-ds/Text2SQL-Flow"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/debugger123/SQLFlow"}],"link_count":3,"sections":9},{"id":"tgpo-2025","title":"TGPO: Tree-Guided Preference Optimization for Robust Web Agent Reinforcement Learning","year":2025,"venue":"ICASSP 2026","authors":["Ziyuan Chen","Zhenghui Zhao","Zhangye Han","Miancan Liu","Xianhang Ye","Yiqing Li","Hongbo Min","Jinkui Ren","Xiantao Zhang","Guitao Cao"],"authors_zh":"Ziyuan Chen、Zhenghui Zhao、Zhangye Han、Miancan Liu、Xianhang Ye、Yiqing Li、Hongbo Min、Jinkui Ren、Xiantao Zhang、Guitao Cao","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","verifier_reward","process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","state_action_level","process_reward","pairwise_preference"],"training_use":["sft","preference_learning","agent_training","evaluation"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["web_agents","browser_automation","ecommerce","gui_interaction"],"tags":["tgpo","web-agent","offline-agent-rl","tree-structured-trajectories","process-reward","state-action-preferences","dynamic-preference-weighting","browser-use","seeact","online-mind2web","c-webshop","release-audit"],"status":"partial","priority":"可读","paper_type_zh":"Web agent 轨迹构造与偏好优化方法论文","best_for_zh":"研究 environment/agent trajectory data、process reward、偏好学习和 live-web 复现审计的读者","confidence":"medium","one_line":["TGPO merges repeated web-agent episodes into state trees and derives mixed process rewards plus weighted action preferences for offline training, but releases no trajectories, trees, verifier decisions, split manifest, environment snapshot, code, or license.","TGPO 把重复 web-agent episode 合并成 trajectory tree，并生成四项 process reward 与 node-level chosen/rejected action pair；机制可用于研究离线 agent 训练，但未核验到代码、数据、模型、环境快照与许可证的公开发布。"],"why":"It shows how terminal web-task outcomes can be converted into state-action supervision without manual step labels, while exposing a central audit boundary: heuristic state merging and an undisclosed VLM judge can silently determine the preference data, and current official artifacts do not support replay or safe training reuse.","primary_link":"https://arxiv.org/pdf/2509.14172v2","links":[],"link_count":4,"sections":9},{"id":"apex-productivity-index-2025","title":"The AI Productivity Index (APEX)","year":2025,"venue":"arXiv","authors":["Bertie Vidgen","Abby Fennelly","Evan Pinnix","Julien Benchek","Daniyal Khan","Zach Richards","Austin Bridges","Calix Huang","Kanishka Sahu","Abhishek Kottamasu","Bo Ma","Ben Hunsberger","Isaac Robinson","Akul Datta","Chirag Mahapatra","Dominic Barton","Cass R. Sunstein","Eric Topol","Brendan Foody","Osvald Nitski"],"authors_zh":"Bertie Vidgen, Abby Fennelly, Evan Pinnix, Julien Benchek, Daniyal Khan 等","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["evaluation","reward_modeling","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["finance","law","consulting","medicine"],"tags":["professional_work","rubric","evaluation","productivity"],"status":"verified","priority":"可读","paper_type_zh":"专业知识工作的人类专家 rubric 评测基准","best_for_zh":"研究专业任务评测、LLM Judge 与经济价值导向能力测量的读者。","confidence":"high","one_line":["APEX assesses professional finance, law, consulting, and medical work using expert-authored cases and rubric grading.","APEX 以资深从业者设计的任务、材料和 rubric，检验模型能否完成金融、法律、咨询和医疗中的高价值专业工作。"],"why":"It tests whether model outputs meet client-ready standards in high-value professional tasks.","primary_link":"https://arxiv.org/abs/2509.25721","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Mercor-Intelligence/apex-evals/tree/main/apex-evals-v1-extended"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mercor/APEX-v1-extended"},{"key":"project","label":["Project","项目主页"],"url":"https://www.mercor.com/apex/"}],"link_count":5,"sections":9},{"id":"art-scaling-test-time-compute-2025","title":"The Art of Scaling Test-Time Compute for Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Aradhye Agarwal","Ayan Sengupta","Tanmoy Chakraborty"],"authors_zh":"Aradhye Agarwal（微软研究院）；Ayan Sengupta、Tanmoy Chakraborty（印度理工学院德里分校）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-scaling","strategy-selection","reasoning-horizon"],"status":"verified","priority":"可读","paper_type_zh":"大规模测试时扩展比较研究","best_for_zh":"需要为异质推理模型选择推理策略的读者。","confidence":"high","one_line":["This study maps when shortest-trace voting, beam search, or majority voting is the best use of test-time compute.","该研究说明最短轨迹投票、束搜索和多数投票各自适用的模型类型、题目难度与测试时预算不同。"],"why":"It argues that no universal test-time strategy exists; the right policy depends on model horizon and compute budget.","primary_link":"https://arxiv.org/abs/2512.02008","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Aradhye2002/art_of_tts"}],"link_count":2,"sections":9},{"id":"biggen-bench-fine-grained-evaluation-2024","title":"The BiGGen Bench: A Principled Benchmark for Fine-grained Evaluation of Language Models with Language Models","year":2025,"venue":"arXiv preprint","authors":["Seungone Kim","Juyoung Suk","Ji Yong Cho","Shayne Longpre","Chaeeun Kim","Dongkeun Yoon","Guijin Son","Yejin Cho","Sheikh Shafayat","Jinheon Baek","Sue Hyun Park","Hyeonbin Hwang","Jinkyung Jo","Hyowon Cho","Haebin Shin","Seongyun Lee","Hanseok Oh","Noah Lee","Namgyu Ho","Se June Joo","Miyoung Ko","Yoonjoo Lee","Hyungjoo Chae","Jamin Shin","Joel Jang","Seonghyeon Ye","Bill Yuchen Lin","Sean Welleck","Graham Neubig","Moontae Lee","Kyungjae Lee","Minjoon Seo"],"authors_zh":"Seungone Kim、Juyoung Suk、Ji Yong Cho、Shayne Longpre、Chaeeun Kim、Dongkeun Yoon、Guijin Son、Yejin Cho、Sheikh Shafayat、Jinheon Baek、Sue Hyun Park、Hyeonbin Hwang、Jinkyung Jo、Hyowon Cho、Haebin Shin、Seongyun Lee、Hanseok Oh、Noah Lee、Namgyu Ho、Se June Joo、Miyoung Ko、Yoonjoo Lee、Hyungjoo Chae、Jamin Shin、Joel Jang、Seonghyeon Ye、Bill Yuchen Lin、Sean Welleck、Graham Neubig、Moontae Lee、Kyungjae Lee、Minjoon Seo","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["BiGGen Bench supplies instance-specific criteria across 77 generation tasks for fine-grained multi-judge evaluation of nine LM capabilities.","BiGGen Bench 为 77 类生成任务提供实例专属标准，以多裁判细粒度测量九种语言模型能力。"],"why":"BiGGen Bench supplies instance-specific criteria across 77 generation tasks for fine-grained multi-judge evaluation of nine LM capabilities.","primary_link":"https://arxiv.org/abs/2406.05761","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/prometheus-eval/prometheus-eval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/prometheus-eval/BiGGen-Bench"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/prometheus-eval/prometheus-eval/tree/main/BiGGen-Bench"}],"link_count":4,"sections":9},{"id":"emperor-contamination-mitigation-2025","title":"The Emperor's New Clothes in Benchmarking? A Rigorous Examination of Mitigation Strategies for LLM Benchmark Data Contamination","year":2025,"venue":"ICLR 2025","authors":["Yifan Sun","Han Wang","Dongbai Li","Gang Wang","Huan Zhang"],"authors_zh":"Yifan Sun, Han Wang, Dongbai Li, Gang Wang, Huan Zhang","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"基准污染缓解策略评测","best_for_zh":"需要选择或设计污染缓解方案的基准维护者。","confidence":"high","one_line":["Official code release evaluates contamination-mitigation strategies under controlled settings.","以保真度和污染抵抗力审计基准更新策略。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=TuvDxubEfE","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ASTRAL-Group/BDC_mitigation_assessment"}],"link_count":2,"sections":9},{"id":"facts-grounding-2025","title":"The FACTS Grounding Leaderboard: Benchmarking LLMs' Ability to Ground Responses to Long-Form Input","year":2025,"venue":"arXiv preprint / Kaggle leaderboard","authors":["Alon Jacovi","Andrew Wang","Chris Alberti","Connie Tao","Jon Lipovetz","Kate Olszewska","Lukas Haas","Michelle Liu","Nate Keating","Adam Bloniarz","Carl Saroufim","Corey Fry","Dror Marcus","Doron Kukliansky","Gaurav Singh Tomar","James Swirhun","Jinwei Xing","Lily Wang","Madhu Gurumurthy","Michael Aaron","Moran Ambar","Rachana Fellinger","Rui Wang","Zizhao Zhang","Sasha Goldshtein","Dipanjan Das"],"authors_zh":"Alon Jacovi 等（Google DeepMind、Google Research、Google Cloud、Kaggle）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["release_audit","reward_verifier_layer"],"domains":["factuality-and-grounding","live-hidden-contamination-audit"],"tags":["benchmark","live_hidden_contamination_audit","factuality-and-grounding"],"status":"verified","priority":"必读","paper_type_zh":"arXiv preprint / Kaggle online benchmark 的 grounding evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["FACTS Grounding exposes long-document grounding with public/private split and judge ensemble as an auditable evaluation surface.","FACTS Grounding 把带公开与私有划分、由评审集成打分的长文档依据核查做成可审计的评测面。"],"why":"Private split plus grounding-specific scoring makes it useful for contamination-resistant evaluation.","primary_link":"https://arxiv.org/abs/2501.03200","links":[{"key":"data","label":["Data","数据"],"url":"https://www.kaggle.com/benchmarks/google/facts-grounding"},{"key":"project","label":["Project","项目主页"],"url":"https://www.kaggle.com/facts-leaderboard"}],"link_count":4,"sections":9},{"id":"lessons-developing-process-reward-models-mathematical-reasoning-2025","title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","year":2025,"venue":"arXiv preprint","authors":["Zhenru Zhang","Chujie Zheng","Yangzhen Wu","Beichen Zhang","Runji Lin","Bowen Yu","Dayiheng Liu","Jingren Zhou","Junyang Lin"],"authors_zh":"Zhenru Zhang、Chujie Zheng、Yangzhen Wu、Beichen Zhang、Runji Lin、Bowen Yu、Dayiheng Liu、Jingren Zhou、Junyang Lin","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","scalar_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","frontier_pipeline","release_audit"],"domains":["mathematics","reasoning"],"tags":["qwen","process-reward-model","process-supervision","mathematical-reasoning","mc-estimation","llm-as-a-judge","consensus-filtering","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"过程奖励模型技术报告与数据披露台账","best_for_zh":"需要审计数学过程监督、MC 标注、LLM critic、consensus filtering 和 PRM 评估偏差的读者","confidence":"medium","one_line":["Qwen process-reward models use MC plus critic consensus step labels and release weights, but not their query, trace, label, calibration, or audit corpus.","Qwen process-reward models 使用 MC 加 critic consensus 的步骤标签并公开权重，但没有发布 queries、traces、labels、calibration 或审计语料。"],"why":"It distinguishes outcome reachability from step verification and makes the data/provenance boundary behind published process-reward weights explicit.","primary_link":"https://arxiv.org/abs/2501.07301","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Qwen/Qwen2.5-Math-PRM-7B"}],"link_count":3,"sections":9},{"id":"llama-4-herd-2025","title":"The Llama 4 Herd: The Beginning of a New Era of Natively Multimodal AI Innovation","year":2025,"venue":"Meta research publication","authors":["unknown"],"authors_zh":"Meta","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["sft","distillation","preference_learning","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["multimodal","general_reasoning"],"tags":["llama-4","meta","frontier-report","disclosure-ledger","multimodal","sft","online-rl","dpo","distillation","model-as-judge"],"status":"partial","priority":"必读","paper_type_zh":"前沿多模态模型技术报告与数据披露账本","best_for_zh":"审计多模态训练与蒸馏披露边界的读者","confidence":"medium","one_line":["Meta reports multimodal pretraining plus hard-example SFT, online RL, DPO, safety tuning, and Behemoth-to-Maverick codistillation for Llama 4, releases Scout/Maverick weights, and withholds the training records and core feedback stack.","Llama 4 Herd 报告了超过 30T 的多模态训练 token、200 种语言，以及 Behemoth 为 Scout 和 Maverick 提供教师能力；但数据清单和后训练反馈均为 unknown。"],"why":"It lets the frontier-report ledger separate released model artifacts from unreleased multimodal data, difficulty labels, policy rollouts, preferences, rewards, and teacher targets, while preventing benchmark results from standing in for data evidence.","primary_link":"https://ai.meta.com/blog/llama-4-multimodal-intelligence/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/meta-llama/llama-models"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/meta-llama/llama-4"}],"link_count":3,"sections":9},{"id":"silent-judge-shortcut-bias-2025","title":"The Silent Judge: Unacknowledged Shortcut Bias in LLM-as-a-Judge","year":2025,"venue":"NeurIPS 2025 Reliable ML Workshop","authors":["Arash Marioriyad","Mohammad Hossein Rohban","Mahdieh Soleymani Baghshah"],"authors_zh":"Arash Marioriyad，Mohammad Hossein Rohban，Mahdieh Soleymani Baghshah。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","paper-page-backed","audit"],"status":"verified","priority":"必读","paper_type_zh":"LLM-as-a-Judge shortcut bias 与理由忠实性审计","best_for_zh":"部署或审计自动成对比较 judge 的研究者和评测团队。","confidence":"medium","one_line":["Tests provenance and recency shortcut biases and the faithfulness of stated judge rationales.","受控注入来源与时间线索，揭示 LLM judge 会改变判决却不在理由中承认该捷径。"],"why":"It adds a paper-backed reliability or failure-mode surface to Track 13.","primary_link":"https://openreview.net/forum?id=6j8jAaDyUG","links":[],"link_count":2,"sections":9},{"id":"toolathlon-2025","title":"The Tool Decathlon: Benchmarking Language Agents for Diverse, Realistic, and Long-Horizon Task Execution","year":2025,"venue":"ICLR 2026","authors":["Junlong Li","Wenshuo Zhao","Jian Zhao","Weihao Zeng","Haoze Wu","Xiaochen Wang","Rui Ge","Yuxuan Cao","Yuzhen Huang","Wei Liu","Junteng Liu","Zhaochen Su","Yiyang Guo","Fan Zhou","Lueyang Zhang","Juan Michelini","Xingyao Wang","Xiang Yue","Shuyan Zhou","Graham Neubig","Junxian He"],"authors_zh":"Junlong Li, Wenshuo Zhao, Jian Zhao, Weihao Zeng, Haoze Wu, Xiaochen Wang, Rui Ge, Yuxuan Cao, Yuzhen Huang, Wei Liu, Junteng Liu, Zhaochen Su, Yiyang Guo, Fan Zhou, Lueyang Zhang, Juan Michelini, Xingyao Wang, Xiang Yue, Shuyan Zhou, Graham Neubig, Junxian He","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","long_horizon_agents","cross_application_workflows"],"tags":["toolathlon","agent-environment","agent-trajectories","mcp","tool-use","long-horizon","execution-based-evaluation","terminal-verifier","benchmark-contamination","do-not-train"],"status":"partial","priority":"可读","paper_type_zh":"长程工具智能体基准、环境与评测轨迹发布","best_for_zh":"研究环境智能体轨迹、MCP 工具调用、终态 verifier、基准漂移与污染审计的读者","confidence":"medium","one_line":["Toolathlon pairs 108 fuzzy cross-application tasks with initialized software states, recorded tool-use episodes, and deterministic terminal evaluators; its gated Verified trajectories are explicitly evaluation/audit-only and must not be used for training.","Toolathlon 将 108 个模糊的跨应用长程任务、初始化软件状态、工具调用 episode 与确定性终态 evaluator 绑定；其 gated Toolathlon-Verified 轨迹仅允许评测与审计，不得用于训练。"],"why":"It makes the entire agent-data contract inspectable--task, tools, observations, environment mutation, terminal state check, run statistics, and failures--while illustrating why environment/evaluator version drift and contamination policy can dominate whether trajectory data is reproducible or reusable.","primary_link":"https://arxiv.org/abs/2510.25726","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/Toolathlon"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hkust-nlp/Toolathlon-Verified_Trajectories"},{"key":"project","label":["Project","项目主页"],"url":"https://toolathlon.xyz/introduction"}],"link_count":8,"sections":9},{"id":"theorem-prover-judge-synthetic-data-2025","title":"Theorem Prover as a Judge for Synthetic Data Generation","year":2025,"venue":"ACL 2025","authors":["Joshua Ong Jun Leang","Giwon Hong","Wenda Li","Shay B. Cohen"],"authors_zh":"Joshua Ong Jun Leang, Giwon Hong, Wenda Li, Shay B. Cohen","tracks":["programmatically_verifiable_outcome_data","data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["step_level","full_episode","pairwise_preference","process_reward"],"training_use":["sft","preference_learning","process_supervision","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematics"],"tags":["theorem-prover-feedback","lean","autoformalisation","synthetic-math-data","step-verification","reverse-question-answering","sft-dpo-routing","formal-verifier","release-unavailable"],"status":"partial","priority":"必读","paper_type_zh":"合成数据构建与定理证明器反馈方案","best_for_zh":"构建合成数学推理数据、集成 Lean 反馈、设计可验证偏好对或审计自动形式化流水线的研究者","confidence":"high","one_line":["A Lean-grounded recipe for generating and selecting synthetic mathematical reasoning, with executable step checks, retry feedback, and explicit SFT/DPO routing.","该工作用 Lean 的可执行证明反馈筛选合成数学推理，并将回答对分配到 SFT、DPO、丢弃或重试路径；核心审计风险是自然语言与形式化陈述的对齐，以及当前不可访问的发布仓库。"],"why":"It makes an executable theorem prover part of the data-allocation contract while exposing formalisation fidelity, selection bias, and unavailable release artifacts as audit risks.","primary_link":"https://aclanthology.org/2025.acl-long.1448/","links":[],"link_count":3,"sections":9},{"id":"numerical-claim-verification-tts-2025","title":"Think Right, Not More: Test-Time Scaling for Numerical Claim Verification","year":2025,"venue":"Findings of EMNLP 2025","authors":["Primakov Chungkham","Venktesh V","Vinay Setty","Avishek Anand"],"authors_zh":"Primakov Chungkham、Venktesh V、Vinay Setty、Avishek Anand（机构：TU Delft、Stockholm University、University of Stavanger）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report"],"domains":["fact-checking","numerical-reasoning"],"tags":["test-time-compute","fact-checking","process-verifier","adaptive-budget","numerical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"自适应验证器引导的测试时扩展研究（Findings of EMNLP 2025）","best_for_zh":"为证据支撑的非数学决策分配推理预算的读者。","confidence":"high","one_line":["VerifierFC uses claim complexity to decide when to sample multiple fact-checking paths and a process verifier to select among them.","VerifierFC 按声明复杂度决定何时采样多条核查路径，并用过程验证器从中选择。"],"why":"It shows that selective expansion can outperform uniform expansion when verification difficulty varies sharply by instance.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1322/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/VenkteshV/VerifierFC"}],"link_count":3,"sections":9},{"id":"think-rm-long-horizon-generative-reward-models-2025","title":"Think-RM: Enabling Long-Horizon Reasoning in Generative Reward Models","year":2025,"venue":"NeurIPS 2025","authors":["Ilgee Hong","Changlong Yu","Liang Qiu","Weixiang Yan","Zhenghao Xu","Haoming Jiang","Qingru Zhang","Qin Lu","Xin Liu","Chao Zhang","Tuo Zhao"],"authors_zh":"Ilgee Hong, Changlong Yu, Liang Qiu, Weixiang Yan, Zhenghao Xu, Haoming Jiang, Qingru Zhang, Qin Lu, Xin Liu, Chao Zhang, Tuo Zhao","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["preference-modeling","reward-modeling","long-horizon-reasoning"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"长程推理生成式奖励模型与偏好评审数据论文","best_for_zh":"需要训练带内部审议过程的生成式奖励模型，或直接利用成对偏好奖励优化策略的研究者。","confidence":"high","one_line":["Think-RM releases long-chain preference judgments that train a generative reward model to deliberate before issuing a pairwise verdict.","6.01K 长 CoT 偏好评审样本，含 chosen/rejected 判定过程，训练能先推理再裁决的通用 GenRM。"],"why":"It exposes preference feedback alongside the reward model's reasoning process and an optimization route that consumes pairwise signals directly.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/fe9f0dcf515c331b124e4f49f760f85e-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/IlgeeHong/Think-RM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ilgee/hs2-naive-reasoning-binary-max"}],"link_count":4,"sections":9},{"id":"thinking-vs-doing-2025","title":"Thinking vs. Doing: Agents that Reason by Scaling Test-Time Interaction","year":2025,"venue":"arXiv preprint","authors":["Junhong Shen","Hao Bai","Lunjun Zhang","Yifei Zhou","Amrith Setlur","Shengbang Tong","Diego Caples","Nan Jiang","Tong Zhang","Ameet Talwalkar","Aviral Kumar"],"authors_zh":"Junhong Shen、Hao Bai、Lunjun Zhang、Yifei Zhou、Amrith Setlur、Shengbang Tong、Diego Caples、Nan Jiang、Tong Zhang、Ameet Talwalkar、Aviral Kumar","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["web-agents","web-navigation","multimodal-agents"],"tags":["environment-agent-trajectory-data","web-agent","online-filtered-bc","interaction-scaling","horizon-curriculum","positive-trajectory-filtering","test-time-compute"],"status":"partial","priority":"必读","paper_type_zh":"网页智能体交互扩展与在线筛选式行为克隆研究","best_for_zh":"关注环境智能体轨迹数据、终局反馈、在线筛选、replay 与交互预算审计的研究者和工程师","confidence":"high","one_line":["TTI turns terminally successful multimodal web episodes into state/action behavior-cloning data under an expanding interaction horizon, but its public release omits trajectories and does not reproduce a clean paper-aligned split/configuration.","TTI 以终局成功筛选在线多模态网页 episode，再对其中的 state-action 步骤做 behavior cloning 并逐步扩大交互 horizon；但公开发布没有 rollout 语料，且划分、judge 与训练配置存在审计缺口。"],"why":"It shows that the interaction budget changes which exploratory behaviors enter post-training data, and it provides a concrete case where terminal feedback, positive-only selection, evaluator identity, split integrity, and replay metadata must be audited together.","primary_link":"https://arxiv.org/abs/2506.07976","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/test-time-interaction/TTI"},{"key":"data","label":["Data","数据"],"url":"https://github.com/test-time-interaction/TTI/tree/92ec04f5e4f2e4e40b99ecb1652191030a54a74a/tasks"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/sjunhongs/tti_webvoyager"},{"key":"project","label":["Project","项目主页"],"url":"https://test-time-interaction.github.io/"}],"link_count":7,"sections":9},{"id":"thinking-vs-doing-interaction-2025","title":"Thinking vs. Doing: Improving Agent Reasoning by Scaling Test-Time Interaction","year":2025,"venue":"NeurIPS 2025","authors":["Junhong Shen","Hao Bai","Lunjun Zhang","Yifei Zhou","Amrith Setlur","Shengbang Tong","Diego Caples","Nan Jiang","Tong Zhang","Ameet S. Talwalkar","Aviral Kumar"],"authors_zh":"Junhong Shen、Hao Bai、Lunjun Zhang、Yifei Zhou、Amrith Setlur、Shengbang Tong、Diego Caples、Nan Jiang、Tong Zhang、Ameet S. Talwalkar、Aviral Kumar（机构：Carnegie Mellon University、Scribe、University of Illinois Urbana-Champaign、University of Toronto、University of California, Berkeley、The AGI Company、New York University）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["environmental"],"supervision_granularity":["full_episode"],"training_use":["test_time_compute","reinforcement_learning","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["web-agents"],"tags":["test-time-compute","agent","web-agent","interaction-horizon","online-rl","exploration"],"status":"verified","priority":"必读","paper_type_zh":"测试时交互扩展与在线强化学习研究（NeurIPS 2025）","best_for_zh":"研究智能体如何在单步思考与环境交互之间分配推理预算的读者。","confidence":"high","one_line":["TTI scales a web agent’s test-time interaction horizon, then trains that behavior with a horizon curriculum so the agent can explore, backtrack, and re-plan.","TTI 将网页智能体的环境交互步数作为测试时预算，并用逐步拉长交互上限的在线学习，使其能够探索、回退和动态重规划。"],"why":"It shows that more test-time actions can acquire information unavailable to a longer reasoning trace before acting.","primary_link":"https://arxiv.org/abs/2506.07976","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/test-time-interaction/TTI"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/sjunhongs/tti_webvoyager"},{"key":"project","label":["Project","项目主页"],"url":"https://test-time-interaction.github.io/"}],"link_count":6,"sections":9},{"id":"thought-anchors-2025","title":"Thought Anchors: Which LLM Reasoning Steps Matter?","year":2025,"venue":"arXiv preprint; submitted to ICLR 2026","authors":["Paul C. Bogdan","Uzay Macar","Neel Nanda","Arthur Conmy"],"authors_zh":"Paul C. Bogdan、Uzay Macar、Neel Nanda、Arthur Conmy","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["data_release","audit_failure","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","release_audit"],"domains":["mathematical_reasoning","general_knowledge_reasoning"],"tags":["thought-anchors","counterfactual-rollouts","sentence-intervention","reasoning-trace-attribution","answer-distribution-feedback","chain-of-thought-interpretability","negative-traces","attention-suppression","receiver-heads","math-rollouts","open-data-release"],"status":"partial","priority":"必读","paper_type_zh":"句子级反事实推理轨迹数据发布与可解释性审计方法","best_for_zh":"研究 rollout 数据结构、CoT 因果归因、负面轨迹、答案分布反馈与大规模发布计数审计的读者","confidence":"high","one_line":["Thought Anchors releases MATH reasoning traces with sentence-level keep/remove continuation banks, correct and incorrect outcomes, answer- distribution importance metrics, and LLM-labeled dependencies for auditing which steps steer downstream reasoning.","Thought Anchors 公开了 MATH 正确与错误推理链、逐句保留/移除的反事实续写、强制回答续写及答案分布指标，用于审计哪些句子改变后续推理；HF 的 20,997 行是原始文件索引而非 rollout 数量。"],"why":"The release turns a reasoning sentence into an auditable intervention unit and preserves downstream continuations rather than only a scalar score, enabling trace attribution and failure analysis; it also demonstrates why file counts, nested rollout counts, generator lineage, and semantic filtering must be reconciled before reusing large rollout corpora.","primary_link":"https://arxiv.org/abs/2506.19143","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/interp-reasoning/thought-anchors"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/uzaymacar/math-rollouts"},{"key":"project","label":["Project","项目主页"],"url":"https://www.thought-anchors.com/"}],"link_count":7,"sections":9},{"id":"thought-calibration-tts-2025","title":"Thought calibration: Efficient and confident test-time scaling","year":2025,"venue":"EMNLP 2025","authors":["Menghua Wu","Cai Zhou","Stephen Bates","Tommi Jaakkola"],"authors_zh":"Menghua Wu、Cai Zhou、Stephen Bates、Tommi Jaakkola（机构：Massachusetts Institute of Technology）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-compute","early-stopping","calibration","hidden-states","reasoning-tokens","efficiency"],"status":"verified","priority":"必读","paper_type_zh":"面向测试时推理的校准早停研究（EMNLP 2025）","best_for_zh":"希望减少推理 token 成本并显式控制提前终止风险的读者。","confidence":"high","one_line":["Thought calibration uses lightweight hidden-state probes and risk control to stop a reasoning trace once further thought is unlikely to add useful work.","Thought calibration 利用隐藏状态上的轻量探针与风险控制，在进一步思考不太可能带来新信息时终止推理轨迹。"],"why":"It treats “when to stop thinking” as a measurable and calibrated test-time budget decision rather than a fixed length heuristic.","primary_link":"https://aclanthology.org/2025.emnlp-main.722/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rmwu/thought-calibration"}],"link_count":3,"sections":9},{"id":"tinyv-verifier-refresh-2025","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","year":2025,"venue":"2nd AI for Math Workshop @ ICML 2025 (Poster)","authors":["Zhangchen Xu","Yuetai Li","Fengqing Jiang","Bhaskar Ramasubramanian","Luyao Niu","Bill Yuchen Lin","Radha Poovendran"],"authors_zh":"Zhangchen Xu、Yuetai Li、Fengqing Jiang、Bhaskar Ramasubramanian、Luyao Niu、Bill Yuchen Lin、Radha Poovendran","tracks":["data_construction_open_release_recipes","audit_failure_contamination_verifier_attacks"],"source_role":["verifier_reward","construction_recipe","data_release","benchmark","model_report","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","scalar_reward"],"training_use":["sft","reward_modeling","rlvr","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["mathematical_reasoning"],"tags":["tinyv","false-negatives","verifier-refresh","learned-verifier","prime-verifier","answer-equivalence","synthetic-augmentation","llm-as-judge","grpo","rlvr","hardverify-math"],"status":"partial","priority":"必读","paper_type_zh":"验证器刷新、开放数据发布与 RLVR 奖励审计","best_for_zh":"研究数学答案等价、Prime 与 learned verifier 的分层奖励、假阴性修复、LLM-as-a-judge 数据构建，以及开放配方的谱系、许可、split 和实现漂移","confidence":"high","one_line":["TinyV refreshes a rule-based verifier by re-judging rejected answers with two large models, augmenting equivalent-answer variants, training a small binary judge, and querying it only after Prime rejection.","TinyV 先让 Prime Verifier 检查答案，仅在其拒绝时调用 1.5B 学习式验证器；公开对象包括 159,136 条平衡 SFT 数据、7,009 条困难提示和 250 条 HardVerify-Math，但 638K 前体池、实验所用 5K 清单、双裁判账本与 held-out 验证器评测未发布。"],"why":"It exposes verifier refresh as a concrete data lifecycle and shows that learned reward recovery must be audited for false positives, split leakage, license, missing lineage, and paper-code drift.","primary_link":"https://openreview.net/pdf?id=scPETXuAiY","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/uw-nsl/TinyV"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zhangchenxu/TinyV_Training_Data_Balanced"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/zhangchenxu/tinyv"}],"link_count":14,"sections":9},{"id":"token-cleaning-sft-2025","title":"Token Cleaning: Fine-Grained Data Selection for LLM Supervised Fine-Tuning","year":2025,"venue":"ICML 2025","authors":["Jinlong Pang","Na Di","Zhaowei Zhu","Jiaheng Wei","Hao Cheng","Chen Qian","Yang Liu"],"authors_zh":"Jinlong Pang、Na Di、Zhaowei Zhu、Jiaheng Wei、Hao Cheng、Chen Qian、Yang Liu","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["unknown"],"supervision_granularity":["step_level"],"training_use":["sft"],"construction_layer":["trace_writing"],"domains":["instruction-tuning"],"tags":["post-training","training-usage"],"status":"verified","priority":"可读","paper_type_zh":"监督微调 Token 级数据筛选论文","best_for_zh":"需要分析 SFT 样本内部监督质量、而非只做整条样本过滤的读者。","confidence":"high","one_line":["Token Cleaning removes low-information supervision tokens according to their estimated influence on model updates.","Token Cleaning 按模型更新对每个 token 的影响筛除低信息监督，以更细粒度方式构造 SFT 训练目标。"],"why":"It makes the link between an explicit data object and a training objective inspectable.","primary_link":"https://proceedings.mlr.press/v267/pang25a.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UCSC-REAL/TokenCleaning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jlpang888/DS2_50k"}],"link_count":4,"sections":9},{"id":"tpo-visual-anchored-rewards-2025","title":"Token Preference Optimization with Self-Calibrated Visual-Anchored Rewards for Hallucination Mitigation","year":2025,"venue":"Findings of EMNLP 2025","authors":["Jihao Gu","Yingyao Wang","Meng Cao","Pi Bu","Jun Song","Yancheng He","Shilong Li","Bo Zheng"],"authors_zh":"Jihao Gu, Yingyao Wang, Meng Cao, Pi Bu, Jun Song, Yancheng He, Shilong Li, Bo Zheng","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["multimodal","alignment"],"tags":["multimodal","token-reward","preference-optimization","hallucination","visual-grounding"],"status":"verified","priority":"必读","paper_type_zh":"视觉落地的 token 级偏好优化研究","best_for_zh":"使用既有偏好对训练视觉语言模型、但没有可扩展 token 级幻觉标注的读者。","confidence":"high","one_line":["TPO dynamically rewards preference-pair tokens by how much their predictions depend on the original image rather than a corrupted image.","TPO 按 token 预测对原图而非扰动图的依赖程度，动态加权既有偏好对中的训练奖励。"],"why":"It makes visual grounding a dynamic, per-token training reward instead of assigning the same preference signal to every response token.","primary_link":"https://aclanthology.org/2025.findings-emnlp.1076/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/alibaba/TPO"}],"link_count":3,"sections":9},{"id":"token-squeeze-2025","title":"TokenSqueeze: Performance-Preserving Compression for Reasoning LLMs","year":2025,"venue":"NeurIPS 2025","authors":["Yuxiang Zhang","Zhengxu Yu","Weihang Pan","Zhongming Jin","Qiang Fu","Deng Cai","Binbin Lin","Jieping Ye"],"authors_zh":"Yuxiang Zhang、Zhengxu Yu、Weihang Pan、Zhongming Jin、Qiang Fu、Deng Cai、Binbin Lin、Jieping Ye","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematical_reasoning","code_reasoning"],"tags":["long2short","reasoning-compression","self-generated-data","preference-pairs","adaptive-depth-selection","KL-constrained-rewriting","dpo","sft","mathematical-reasoning","rollout-selection"],"status":"partial","priority":"可读","paper_type_zh":"推理轨迹压缩与偏好数据构造方法","best_for_zh":"需要审计 Long2Short、rollout 筛选、答案级 verifier、偏好对与长度正则化目标的研究者","confidence":"medium","one_line":["TokenSqueeze self-samples answer-checked traces, selects depth-adaptive correct positives against longer incorrect responses, KL-filters step rewrites, and trains concise reasoners with SFT plus length-aware DPO; the final generated corpus remains unreleased.","TokenSqueeze 用自采样、答案核验的推理轨迹构造自适应深度的正例与更长的错误负例，再以 KL 约束筛选逐步改写并结合 SFT 与长度感知 DPO 训练短推理；最终生成语料尚未发布，因而只能审计其配方而非逐条数据。"],"why":"It makes long-to-short reasoning a data-construction and feedback-contract problem: its efficiency depends on rollout budget, answer extraction, selector, negative-pair rule, KL approximation, and objective. Missing lineage, split, license, generated records, and decontamination evidence block data-level reuse claims.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/file/8442d394d8266db6fdf92583a4783e8c-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zhangyx1122/TokenSqueeze"},{"key":"data","label":["Data","数据"],"url":"https://github.com/zhangyx1122/TokenSqueeze/blob/db674a53650266d7baea483e75695892a2430b2c/datasets/math14k.jsonl"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zhangyx/TokenSqueeze-7B"}],"link_count":8,"sections":9},{"id":"tool-zero-2025","title":"Tool Zero: Training Tool-Augmented LLMs via Pure RL from Scratch","year":2025,"venue":"Findings of the Association for Computational Linguistics: EMNLP 2025","authors":["Yirong Zeng","Xiao Ding","Yutai Hou","Yuxian Wang","Li Du","Juyi Dai","Qiuyang Ding","Duyu Tang","Dandan Tu","Weiwen Liu","Bing Qin","Ting Liu"],"authors_zh":"Yirong Zeng, Xiao Ding, Yutai Hou, Yuxian Wang, Li Du, Juyi Dai, Qiuyang Ding, Duyu Tang, Dandan Tu, Weiwen Liu, Bing Qin, Ting Liu","tracks":["environment_agent_trajectory_data","training_usage_optimization_objectives"],"source_role":["model_report","verifier_reward","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","tool_use","function_calling","tool_integrated_reasoning"],"tags":["environment-agent-trajectory-data","training-usage-optimization-objectives","tool-use","function-calling","pure-rl","grpo","rule-based-reward","reward-scheduling","tool-generalization","offline-reference-verification"],"status":"partial","priority":"可读","paper_type_zh":"函数调用 RLVR 配方与规则奖励研究","best_for_zh":"研究工具调用轨迹、程序化 verifier、GRPO 奖励设计与发布审计的读者","confidence":"medium","one_line":["Tool Zero applies pure GRPO to offline ToolACE/xLAM function-calling records and dynamically tightens each completion's scalar rule reward from exploratory token overlap to strict reference-AST correctness.","Tool Zero 在离线 ToolACE/xLAM 函数调用记录上用 GG-GRPO 将整段 completion 的规则奖励从参考 token 重叠逐步收紧到 AST 等价；它展示了无 SFT 的 RLVR 配方，但专属代码、处理后语料与 checkpoint 均未发布。"],"why":"It is a concrete example of how trajectory-shaped tool-use data can support RLVR without SFT, while showing that 'pure RL' still depends on reference calls and that unreleased processed data, reward code, checkpoints, benchmark pins, and reporting inconsistencies materially limit reuse.","primary_link":"https://aclanthology.org/2025.findings-emnlp.485/","links":[],"link_count":12,"sections":9},{"id":"toolhop-a-query-driven-benchmark-for-evaluating-large-language-models-in-multi-hop-tool-","title":"ToolHop: A Query-Driven Benchmark for Evaluating Large Language Models in Multi-Hop Tool Use","year":2025,"venue":"ACL 2025","authors":["Junjie Ye","Zhengyin Du","Xuesong Yao","Weijian Lin","Yufei Xu","Zehui Chen","Zaiyuan Wang","Sining Zhu","Zhiheng Xi","Siyu Yuan","Tao Gui","Qi Zhang","Xuanjing Huang","Jiecao Chen"],"authors_zh":"Junjie Ye, Zhengyin Du, Xuesong Yao, Weijian Lin, Yufei Xu, Zehui Chen, Zaiyuan Wang, Sining Zhu, Zhiheng Xi, Siyu Yuan, Tao Gui, Qi Zhang, Xuanjing Huang, Jiecao Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["reasoning"],"tags":["programmatic-verification"],"status":"verified","priority":"可读","paper_type_zh":"可程序化验证的推理数据或基准","best_for_zh":"需要可复现结果验证的推理数据研究者。","confidence":"medium","one_line":["One-sentence position:** A verifiable multi-hop tool-use benchmark with 995 queries and 3,912 local tools.","995 个查询与 3,912 个本地工具构成可验证的多跳工具使用 benchmark。"],"why":"It exposes a reproducible terminal outcome check rather than only a textual reference.","primary_link":"https://arxiv.org/abs/2501.02506","links":[{"key":"code","label":["Code","代码"],"url":"https://huggingface.co/datasets/bytedance-research/ToolHop"}],"link_count":4,"sections":9},{"id":"toolmind-2025","title":"ToolMind Technical Report: A Large-Scale, Reasoning-Enhanced Tool-Use Dataset","year":2025,"venue":"arXiv preprint (2025)","authors":["Chen Yang","Ran Le","Yun Xing","Zhenwei An","Zongchao Chen","Wayne Xin Zhao","Yang Song","Tao Zhang"],"authors_zh":"Chen Yang、Ran Le、Yun Xing、Zhenwei An、Zongchao Chen、Wayne Xin Zhao、Yang Song、Tao Zhang","tracks":["instruction_demonstration_rationale_data","process_trace_supervision_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English function calling and multi-turn tool use"],"tags":["instruction-demonstration-rationale","arxiv-2511.15718","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"工具使用监督微调","confidence":"high","one_line":["ToolMind constructs a function graph, simulates realistic interactions, and filters each turn before retaining complete self-corrective trajectories.","ToolMind 用函数图驱动多代理交互，并在逐轮过滤后发布 16 万条合成与 20 万条增强工具记录。"],"why":"Trajectory-level acceptance can hide a wrong intermediate tool call whose error propagates through every later turn.","primary_link":"https://arxiv.org/abs/2511.15718","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Nanbeige/ToolMind"}],"link_count":3,"sections":9},{"id":"toolscale-2025","title":"ToolOrchestra: Elevating Intelligence via Efficient Model and Tool Orchestration","year":2025,"venue":"arXiv preprint (2025)","authors":["Hongjin Su","Shizhe Diao","Ximing Lu","Mingjie Liu","Jiacheng Xu","Xin Dong","Yonggan Fu","Peter Belcak","Hanrong Ye","Hongxu Yin","Yi Dong","Evelina Bakhturina","Tao Yu","Yejin Choi","Jan Kautz","Pavlo Molchanov"],"authors_zh":"Hongjin Su、Shizhe Diao、Ximing Lu、Mingjie Liu、Jiacheng Xu、Xin Dong、Yonggan Fu、Peter Belcak、Hanrong Ye、Hongxu Yin、Yi Dong、Evelina Bakhturina、Tao Yu、Yejin Choi、Jan Kautz、Pavlo Molchanov","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["English model routing, web search, code execution, and multi-tool orchestration"],"tags":["instruction-demonstration-rationale","arxiv-2511.21689","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"强化学习与任务条件化工具编排训练","confidence":"high","one_line":["ToolScale serializes difficult tool-routing scenarios with initial state and evaluation actions, supplying the task substrate for ToolOrchestra training.","ToolScale 把 4063 个困难工具编排任务写成场景、初始状态与评估条件，供小型编排器训练。"],"why":"Small orchestrators need explicit tasks and evaluation criteria to learn when to invoke expensive models or tools rather than always selecting the strongest option.","primary_link":"https://arxiv.org/abs/2511.21689","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/ToolScale"}],"link_count":2,"sections":9},{"id":"toucan-2025","title":"TOUCAN: Synthesizing 1.5M Tool-Agentic Data from Real-World MCP Environments","year":2025,"venue":"arXiv preprint","authors":["Zhangchen Xu","Adriana Meza Soria","Shawn Tan","Anurag Roy","Ashish Sunil Agrawal","Radha Poovendran","Rameswar Panda"],"authors_zh":"Zhangchen Xu、Adriana Meza Soria、Shawn Tan、Anurag Roy、Ashish Sunil Agrawal、Radha Poovendran、Rameswar Panda","tracks":["environment_agent_trajectory_data","instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","model_context_protocol"],"tags":["toucan","agent-trajectories","mcp","tool-use","synthetic-data","real-tool-execution","multi-turn","llm-as-judge","supervised-fine-tuning","environment-drift"],"status":"partial","priority":"必读","paper_type_zh":"真实 MCP 环境合成工具 agent 轨迹数据发布与 SFT 配方","best_for_zh":"关注工具 agent SFT、MCP 轨迹构造、混合评审契约、环境回放与数据审计的研究者","confidence":"high","one_line":["TOUCAN releases 1,527,259 synthetic task-tool episodes generated through 495 live MCP servers plus a 119,287-row SFT subset, while mixed heuristic/judge labels and mutable remote environments limit correctness guarantees and replay.","TOUCAN 发布 1,527,259 条经 495 个真实 MCP 服务生成的工具 agent 轨迹，并从中筛出 119,287 行 SFT 子集；混合规则与模型评审及可变远程环境限制了任务正确性保证和精确回放。"],"why":"It makes large-scale agent SFT data inspectable from task and tool declarations through calls, observations, filtering labels, and final responses, and shows why real execution, task correctness, historical release fidelity, and executable replay are four different audit claims.","primary_link":"https://arxiv.org/abs/2510.01179","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TheAgentArk/Toucan"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Agent-Ark/Toucan-1.5M"}],"link_count":4,"sections":9},{"id":"towards-alignment-centric-instruction-tuning-survey-2025","title":"Towards Alignment-Centric Paradigm: A Survey of Instruction Tuning in Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Xudong Han","Junjie Yang","Tianyang Wang","Ziqian Bi","Xinyuan Song","Junfeng Hao","Junhao Song"],"authors_zh":"Xudong Han、Junjie Yang、Tianyang Wang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing"],"domains":["instruction-tuning","alignment"],"tags":["foundations-and-primers","instruction-tuning","alignment-taxonomy","arXiv-2025"],"status":"verified","priority":"可读","paper_type_zh":"2025 年指令微调与对齐综述预印本","best_for_zh":"关于专家示范、教师蒸馏与自我改进数据的指令微调综述。","confidence":"medium","one_line":["This survey organizes instruction tuning around data origin, tuning strategy, and evaluation rather than treating all prompts and answers as one data type.","该综述从数据来源、调优策略和评测协议梳理大语言模型指令微调的对齐逻辑。"],"why":"It makes the provenance difference between expert demonstrations, teacher distillation, and self-improvement explicit at the atlas entry point.","primary_link":"https://arxiv.org/abs/2508.17184","links":[],"link_count":1,"sections":9},{"id":"contamination-detection-modern-llms-2025","title":"Towards Data Contamination Detection for Modern Large Language Models: Limitations, Inconsistencies, and Oracle Challenges","year":2025,"venue":"COLING 2025 (Main Conference)","authors":["Vinay Samuel","Yue Zhou","Henry Peng Zou"],"authors_zh":"Vinay Samuel、Yue Zhou、Henry Peng Zou","tracks":["data_construction_open_release_recipes","audit_failure_contamination_verifier_attacks"],"source_role":["benchmark","data_release","construction_recipe","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["audit","evaluation","sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["mathematical_reasoning","code_reasoning","knowledge_qa","reading_comprehension","text_classification"],"tags":["data-contamination","contamination-detection","membership-inference","benchmark-audit","oracle-contamination","instruction-tuning","false-positive","false-negative","release-incomplete"],"status":"partial","priority":"必读","paper_type_zh":"数据污染检测审计、受控构造 recipe 与负结果研究","best_for_zh":"设计基准污染审计、核查检测器假设、构造已知新增暴露对照，或评估开放审计 artifact 完整性的读者","confidence":"high","one_line":["Five prompt-, likelihood-, completion-, and order-based probes are compared on public benchmark slices; none consistently tracks known added fine-tuning exposure, and cross-method agreement is weak.","该研究把五种污染检测器作为不同的审计接口，在八个基准与一个受控 LLaMA-2 指令微调 oracle 上比较；五种指标都未能稳定追踪已知新增暴露比例，方法间一致性也很弱。"],"why":"The study turns detector assumptions into auditable interfaces and supplies a rare known-exposure control, while showing that artifacts, distribution shifts, missing thresholds, and incompatible statistics can masquerade as evidence about memorization.","primary_link":"https://aclanthology.org/2025.coling-main.338/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/vsamuel2003/data-contamination"},{"key":"data","label":["Data","数据"],"url":"https://github.com/vsamuel2003/data-contamination/tree/f8ab237abfa44e67676b48f04d09566b59c5740a/datasets"}],"link_count":8,"sections":9},{"id":"towards-large-reasoning-models-2025","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","year":2025,"venue":"arXiv preprint","authors":["Fengli Xu","Qianyue Hao","Zefang Zong","Jingwei Wang","Yunke Zhang","Jingyi Wang","Xiaochong Lan","Jiahui Gong","Tianjian Ouyang","Fanjin Meng","Chenyang Shao","Yuwei Yan","Qinglong Yang","Yiwen Song","Sijian Ren","Xinyuan Hu","Yu Li","Jie Feng","Chen Gao","Yong Li"],"authors_zh":"Fengli Xu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["unknown"],"construction_layer":["release_audit"],"domains":["reasoning","reinforced-reasoning","reasoning-data"],"tags":["reasoning-survey","reasoning-trajectories","reinforcement-learning","test-time-scaling"],"status":"verified","priority":"必读","paper_type_zh":"大推理模型与强化推理综述","best_for_zh":"涵盖推理轨迹、奖励、搜索与推理预算关系的大推理模型综述。","confidence":"high","one_line":["A 2025 survey of reinforced reasoning that brings automated trajectory construction, learning-to-reason, and test-time scaling into one account.","将强化推理、自动数据构建与推理期扩展纳入同一概念图的综述。"],"why":"It distinguishes reasoning trajectories, reward design, policy learning, and inference-time computation as related but different parts of a large reasoning model.","primary_link":"https://arxiv.org/abs/2501.09686","links":[],"link_count":2,"sections":9},{"id":"huatuogpt-o1-2025","title":"Towards Medical Complex Reasoning with LLMs through Medical Verifiable Problems","year":2025,"venue":"Findings of ACL 2025","authors":["Junying Chen","Zhenyang Cai","Ke Ji","Xidong Wang","Wanlong Liu","Rongsheng Wang","Jianye Hou","Benyou Wang"],"authors_zh":"Junying Chen、Zhenyang Cai、Ke Ji、Xidong Wang、Wanlong Liu、Rongsheng Wang、Jianye Hou、Benyou Wang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["verifiable-complex-medical-reasoning-data"],"tags":["instruction-demonstration-rationale","arxiv-2412.18925","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"医学复杂推理监督微调后接可验证强化学习","confidence":"high","one_line":["HuatuoGPT-o1 filters medical exams into 40K verifiable problems, distills complex traces for half, and uses the remaining half for answer-reward RL.","HuatuoGPT-o1 把医学考试题筛成 4 万道可验证问题，其中一半带复杂推理用于微调，另一半用于强化学习。"],"why":"Open medical chat data lacks hard, objectively checkable problems for teaching complex reasoning and supporting stable reinforcement learning.","primary_link":"https://arxiv.org/abs/2412.18925","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/FreedomIntelligence/medical-o1-reasoning-SFT"}],"link_count":2,"sections":9},{"id":"thinking-optimal-scaling-2025","title":"Towards Thinking-Optimal Scaling of Test-Time Compute for LLM Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Wenkai Yang","Shuming Ma","Yankai Lin","Furu Wei"],"authors_zh":"Wenkai Yang、Shuming Ma、Yankai Lin、Furu Wei（机构：中国人民大学高瓴人工智能学院、Microsoft Research）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","answer_level"],"training_use":["test_time_compute","sft"],"construction_layer":["scaling_report","filtering_curation"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","reasoning-effort","overthinking","chain-of-thought","data-selection"],"status":"verified","priority":"必读","paper_type_zh":"测试时计算分配与自改进研究（NeurIPS 2025）","best_for_zh":"研究推理长度控制、过度思考与按题分配推理预算的读者。","confidence":"high","one_line":["TOPS selects shortest correct effort-conditioned reasoning traces to train task-dependent test-time thinking length.","TOPS 从不同推理努力下的候选中选择最短正确回答，训练模型按题目选择合适的测试时思考长度。"],"why":"It treats reasoning length as an allocation variable that can hurt accuracy when extended indiscriminately.","primary_link":"https://arxiv.org/abs/2502.18080","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCBM/TOPS"}],"link_count":4,"sections":9},{"id":"long-context-swe-rl-2025","title":"Training Long-Context, Multi-Turn Software Engineering Agents with Reinforcement Learning","year":2025,"venue":"ICLR 2026 submission","authors":["Alexander Golubev","Maria Trofimova","Sergei Polezhaev","Ibragim Badertdinov","Maksim Nekrashevich","Anton Shevtsov","Simon Karasik","Sergey Abramov","Andrei Andriushchenko","Filipp Fisin","Sergei Skvortsov","Boris Yangel"],"authors_zh":"Alexander Golubev、Maria Trofimova、Sergei Polezhaev、Ibragim Badertdinov、Maksim Nekrashevich、Anton Shevtsov、Simon Karasik、Sergey Abramov、Andrei Andriushchenko、Filipp Fisin、Sergei Skvortsov、Boris Yangel","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["software_engineering","repository_agents","agent_trajectories","long_context","environment_interaction"],"tags":["software-engineering-agents","repository-repair","agent-trajectories","long-context","rejection-fine-tuning","dapo","environment-reward","validation-tests","success-only-selection","sparse-terminal-reward","replay-risk","release-gap","version-drift"],"status":"partial","priority":"必读","paper_type_zh":"软件工程智能体强化学习与轨迹构造方法论文","best_for_zh":"研究环境交互轨迹、软件工程 agent、RFT/RLVR、终局验证器、长上下文训练与可重放性审计的读者","confidence":"high","one_line":["The paper filters 21,336 SWE-rebench tasks to 7,249, keeps 6,548 test-passing self-generated episodes for RFT, then applies synchronous DAPO to 65k/131k repository interactions, but does not release the trajectories, training code, or checkpoint.","论文从 21,336 个 SWE-rebench 任务筛出 7,249 个任务，以测试通过的 6,548 条自生成 episode 做 RFT，再以终局测试奖励训练长上下文 DAPO agent；其训练轨迹、代码与模型均未发布。"],"why":"It specifies how repository state, tool observations, test-based terminal reward, success-only warm-up data, long-context rollout groups, and failure retention interact in SWE-agent post-training, while showing why upstream task containers alone do not make the paper's learning data reusable or replayable.","primary_link":"https://arxiv.org/abs/2508.03501","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nebius/SWE-rebench"}],"link_count":10,"sections":9},{"id":"swe-gym-2025","title":"Training Software Engineering Agents and Verifiers with SWE-Gym","year":2025,"venue":"ICML 2025","authors":["Jiayi Pan","Xingyao Wang","Graham Neubig","Navdeep Jaitly","Heng Ji","Alane Suhr","Yizhe Zhang"],"authors_zh":"Jiayi Pan、Xingyao Wang、Graham Neubig、Navdeep Jaitly、Heng Ji、Alane Suhr、Yizhe Zhang","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","agent_environment","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["state_action_level","full_episode","scalar_reward","trajectory_value"],"training_use":["sft","reward_modeling","agent_training","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["software_engineering","repository_level_code","agentic_reasoning"],"tags":["swe-gym","software-engineering-agents","repository-level-tasks","executable-environments","unit-test-verification","agent-trajectories","rejection-sampling","outcome-reward-model","best-of-n","open-data","failed-trajectories","immutable-manifest-risk","license-drift"],"status":"partial","priority":"必读","paper_type_zh":"软件工程 agent 环境、轨迹数据与 outcome verifier 开放发布","best_for_zh":"研究 repository-level agent SFT、执行反馈、失败轨迹、outcome reward model、Best@k，以及容器、谱系、许可和版本漂移","confidence":"high","one_line":["SWE-Gym releases 2,438 unit-test-validated Python repository tasks plus OpenHands and Moatless trajectories, policies, and outcome verifiers built through rejection sampling and Best@k selection.","SWE-Gym 发布 2,438 个带仓库快照、容器环境和单元测试的 Python 工程任务，并把 491 条成功轨迹、5,564 条失败轨迹和 1,318/1,318 平衡 verifier 数据分开封装；但 Lite 数量、镜像 digest、轨迹谱系和各工件许可仍不一致。"],"why":"It connects repository snapshots, tests, full agent episodes, success/failure selection, and learned trajectory rewards while exposing provenance, license, failure-retention, and image-pinning risks.","primary_link":"https://proceedings.mlr.press/v267/pan25g.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-Gym/SWE-Gym"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/SWE-Gym/SWE-Gym"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/SWE-Gym"},{"key":"project","label":["Project","项目主页"],"url":"https://machinelearning.apple.com/research/training-software"}],"link_count":20,"sections":9},{"id":"training-vision-language-process-reward-models-test-time-scaling-2025","title":"Training Vision-Language Process Reward Models for Test-Time Scaling in Multimodal Reasoning: Key Insights and Lessons Learned","year":2025,"venue":"arXiv","authors":["Brandon Ong","Tej Deep Pala","Vernon Toh","William Chandra Tjhi","Soujanya Poria"],"authors_zh":"Brandon Ong、Tej Deep Pala、Vernon Toh、William Chandra Tjhi、Soujanya Poria","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal","vision_language","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"多模态过程监督数据集与测试时扩展研究","best_for_zh":"训练视觉语言过程奖励模型，或研究多模态推理的测试时筛选。","confidence":"high","one_line":["VL-PRM300K provides roughly 300K multimodal reasoning traces with visual-versus-reasoning error locations for training process reward models.","VL-PRM300K 提供约 30 万条多模态推理轨迹，并标明视觉或推理错误位置，用于训练过程奖励模型。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2509.23250","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/theogbrand/vlprm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ob11/VL-PRM300K"}],"link_count":3,"sections":9},{"id":"tree-rl-2025","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","year":2025,"venue":"ACL 2025","authors":["Zhenyu Hou","Ziniu Hu","Yujiang Li","Rui Lu","Jie Tang","Yuxiao Dong"],"authors_zh":"Zhenyu Hou, Ziniu Hu, Yujiang Li, Rui Lu, Jie Tang, Yuxiao Dong","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","process_supervision"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","trajectory_value"],"training_use":["process_supervision","rlvr"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","code","reasoning"],"tags":["on-policy-rl","tree-search","entropy-guided-search","process-supervision","reasoning-traces"],"status":"partial","priority":"可读","paper_type_zh":"基于熵引导树搜索的 on-policy 强化学习与过程监督配方","best_for_zh":"需要审计推理搜索轨迹、过程奖励、终点 verifier 与训练预算归因的读者","confidence":"medium","one_line":["TreeRL uses entropy-guided on-policy search trees to turn terminal correctness into reweighted step-level rewards; the public repository releases input prompts and code, not a verified corpus of the resulting runtime trees.","TreeRL 将熵引导的 on-policy 搜索树中叶节点的终点正确性转化为重加权逐步奖励；公开仓库发布了输入提示与代码，但未核验发布运行时树语料。"],"why":"For Rollout, Search, and Test-Time Trace Data, TreeRL shows that a trace-data contract includes search topology, budget, terminal checking, and selection—not merely a final answer—and that missing raw trees block full audit.","primary_link":"https://aclanthology.org/2025.acl-long.604/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/TreeRL"},{"key":"data","label":["Data","数据"],"url":"https://github.com/THUDM/TreeRL/blob/main/datasets/train/train_30k.jsonl"}],"link_count":6,"sections":9},{"id":"tritonbench-triton-operators-2025","title":"TritonBench: Benchmarking Large Language Model Capabilities for Generating Triton Operators","year":2025,"venue":"Findings of ACL 2025","authors":["Jianling Li","Shangzhan Li","Zhenye Gao","Qi Shi","Yuxuan Li","Zefan Wang","Jiacheng Huang","Haojie Wang","Jianrong Wang","Xu Han","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Jianling Li, Shangzhan Li, Zhenye Gao, Qi Shi, Yuxuan Li, Zefan Wang, Jiacheng Huang, Haojie Wang, Jianrong Wang, Xu Han, Zhiyuan Liu, Maosong Sun","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["gpu-programming","code-generation","triton"],"tags":["triton","gpu-kernels","profiling","code-generation","2025"],"status":"verified","priority":"可读","paper_type_zh":"Triton GPU 算子生成的双重程序化评测基准","best_for_zh":"需要评测 Triton 代码生成、性能优化或 GPU 算子正确性的研究者。","confidence":"high","one_line":["TritonBench evaluates generated Triton operators through 184 real-world kernels plus GPU correctness and efficiency profiling.","TritonBench 覆盖 184 个真实 Triton 算子，并用 GPU 执行正确性与性能剖析共同评测生成代码。"],"why":"It distinguishes executable correctness from hardware-level efficiency instead of conflating them.","primary_link":"https://aclanthology.org/2025.findings-acl.1183/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thunlp/TritonBench"}],"link_count":4,"sections":9},{"id":"ui-genie-2025","title":"UI-Genie: A Self-Improving Approach for Iteratively Boosting MLLM-based Mobile GUI Agents","year":2025,"venue":"NeurIPS 2025","authors":["Han Xiao","Guozhi Wang","Yuxiang Chai","Zimu Lu","Weifeng Lin","Hao He","Lue Fan","Liuyang Bian","Rui Hu","Liang Liu","Shuai Ren","Yafei Wen","Xiaoxin Chen","Aojun Zhou","Hongsheng Li"],"authors_zh":"Han Xiao、Guozhi Wang、Yuxiang Chai、Zimu Lu、Weifeng Lin、Hao He、Lue Fan、Liuyang Bian、Rui Hu、Liang Liu、Shuai Ren、Yafei Wen、Xiaoxin Chen、Aojun Zhou、Hongsheng Li","tracks":["data_construction_open_release_recipes","preference_reward_feedback_data"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["step_level","full_episode","scalar_reward"],"training_use":["sft","reward_modeling","agent_training"],"construction_layer":["prompt_sourcing","self_play_anchor","reward_verifier_layer","frontier_pipeline"],"domains":["mobile_gui","multimodal_agents"],"tags":["ui-genie","gui-agents","mobile-agents","trajectory-corruption","hard-negative-mining","reward-model-data","iterative-self-improvement"],"status":"partial","priority":"必读","paper_type_zh":"移动 GUI 奖励数据与迭代式 agent 数据构造配方","best_for_zh":"关注多模态 GUI agent、奖励模型、轨迹筛选、自改进数据闭环及发布审计的研究者","confidence":"high","one_line":["Builds mobile-GUI reward data from rule checks, controlled corruptions, and mined hard negatives, then uses a learned verifier to collect step-level agent SFT data over three rounds.","以规则核验、受控破坏和困难负例构造移动 GUI 奖励数据，再用学习到的验证器在三轮探索中筛选步级 agent SFT 数据。"],"why":"Makes an iterative verifier-guided mobile-agent data engine inspectable while exposing count drift, verifier feedback loops, privacy, licensing, and replay risks.","primary_link":"https://arxiv.org/abs/2505.21496","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Euphoria16/UI-Genie"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HanXiao1999/UI-Genie-Agent-16k"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/HanXiao1999/UI-Genie-RM-517k"}],"link_count":7,"sections":9},{"id":"ui-tars-2-2025","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","year":2025,"venue":"arXiv preprint","authors":["Haoming Wang","Haoyang Zou","Huatong Song","Jiazhan Feng","Junjie Fang","Junting Lu","Longxiang Liu","Qinyu Luo","Shihao Liang","Shijue Huang","Wanjun Zhong","Yining Ye","Yujia Qin","Yuwen Xiong","Yuxin Song","Zhiyong Wu","Aoyan Li","Bo Li","Chen Dun","Chong Liu","Daoguang Zan","Fuxing Leng","Hanbin Wang","Hao Yu","Haobin Chen","Hongyi Guo","Jing Su","Jingjia Huang","Kai Shen","Kaiyu Shi","Lin Yan","Peiyao Zhao","Pengfei Liu","Qinghao Ye","Renjie Zheng","Shulin Xin","Wayne Xin Zhao","Wen Heng","Wenhao Huang","Wenqian Wang","Xiaobo Qin","Yi Lin","Youbin Wu","Zehui Chen","Zihao Wang","Baoquan Zhong","Xinchun Zhang","Xujing Li","Yuanfan Li","Zhongkai Zhao","Chengquan Jiang","Faming Wu","Haotian Zhou","Jinlin Pang","Li Han","Qi Liu","Qianli Ma","Siyao Liu","Songhua Cai","Wenqi Fu","Xin Liu","Yaohui Wang","Zhi Zhang","Bo Zhou","Guoliang Li","Jiajun Shi","Jiale Yang","Jie Tang","Li Li","Qihua Han","Taoran Lu","Woyu Lin","Xiaokang Tong","Xinyao Li","Yichi Zhang","Yu Miao","Zhengxuan Jiang","Zili Li","Ziyuan Zhao","Chenxin Li","Dehua Ma","Feng Lin","Ge Zhang","Haihua Yang","Hangyu Guo","Hongda Zhu","Jiaheng Liu","Junda Du","Kai Cai","Kuanye Li","Lichen Yuan","Meilan Han","Minchao Wang","Shuyue Guo","Tianhao Cheng","Xiaobo Ma","Xiaojun Xiao","Xiaolong Huang","Xinjie Chen","Yidi Du","Yilin Chen","Yiwen Wang","Zhaojian Li","Zhenzhu Yang","Zhiyuan Zeng","Chaolin Jin","Chen Li","Hao Chen","Haoli Chen","Jian Chen","Qinghao Zhao","Guang Shi"],"authors_zh":"Haoming Wang 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","agent_environment","benchmark"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","reward_modeling"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["gui-agents","software-environments","web-browsing","mobile","games","tool-use"],"tags":["frontier-report","gui-agent","multi-turn-rl","data-flywheel","sandbox-rollouts","outcome-reward-model","disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"GUI 智能体技术报告","best_for_zh":"研究智能体环境、rollout 与后训练披露的读者","confidence":"medium","one_line":["UI-TARS-2 discloses a continual-pre-training/SFT/PPO flywheel over stateful GUI episodes and three outcome-feedback paths, while leaving the training corpus, reward records, model weights, and replayable environments unreleased.","UI-TARS-2 披露数据飞轮、多轮 RL 与大规模 rollout 沙箱，但未披露训练记录和奖励账本。"],"why":"For the frontier-report disclosure ledger, it is a strong example of method-level transparency without item-level data or environment reproducibility: the episode contract can be reconstructed conceptually, but the claimed pipeline cannot be independently audited end to end.","primary_link":"https://arxiv.org/abs/2509.02544","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/bytedance/UI-TARS"},{"key":"project","label":["Project","项目主页"],"url":"https://seed-tars.com/showcase/ui-tars-2"}],"link_count":4,"sections":9},{"id":"tar-swe-trajectories-2025","title":"Understanding Software Engineering Agents: A Study of Thought-Action-Result Trajectories","year":2025,"venue":"ASE 2025","authors":["Islem Bouzenia","Michael Pradel"],"authors_zh":"Islem Bouzenia、Michael Pradel","tracks":["environment_agent_trajectory_data"],"source_role":["data_release","audit_failure","construction_recipe"],"verification_contract":["mixed","judgment_required"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["trace_writing","release_audit"],"domains":["software_engineering","automated_program_repair","repository_level_code","agent_trajectories"],"tags":["software-engineering-agents","thought-action-result","trajectory-analysis","automated-program-repair","failure-analysis","manual-annotation","open-data"],"status":"partial","priority":"可读","paper_type_zh":"软件工程 agent 轨迹数据发布与混合方法实证研究","best_for_zh":"研究 agent trajectory、软件修复行为、反馈利用、失败模式和轨迹发布审计的读者","confidence":"high","one_line":["Releases and analyzes 120 RepairAgent, AutoCodeRover, and OpenHands episodes as 2,822 thought-action-result iterations with action and semantic-relation labels, while replay metadata and an immutable release remain incomplete.","该研究把 3 个软件工程 agent 的 120 条成功/失败运行统一为 2,822 次 thought-action-result 迭代，并加入动作与语义关系标注；它适合评测与审计，但缺少可重放环境和不可变发布。"],"why":"The work makes agent behavior auditable below the final patch by separating tool feedback, human semantic judgments, and episode success; it is evidence for evaluation and failure analysis, not evidence of a validated training recipe.","primary_link":"https://doi.org/10.1109/ASE63991.2025.00234","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sola-st/llm-agents-study"},{"key":"project","label":["Project","项目主页"],"url":"https://cispa.de/en/research/publications/104312-understanding-software-engineering-agents-a-study-of-thought-action-result-trajectories"}],"link_count":5,"sections":9},{"id":"schoenfeld-episode-theory-lrm-2025","title":"Understanding the Thinking Process of Reasoning Models: A Perspective from Schoenfeld's Episode Theory","year":2025,"venue":"NeurIPS 2025","authors":["Ming Li","Nan Zhang","Chenrui Fan","Hong Jiao","Yanbin Fu","Sydney Peters","Qingshu Xu","Robert Lissitz","Tianyi Zhou"],"authors_zh":"Ming Li、Nan Zhang、Chenrui Fan、Hong Jiao、Yanbin Fu、Sydney Peters、Qingshu Xu、Robert Lissitz、Tianyi Zhou","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reasoning-models","cognitive-episodes","math-traces"],"tags":["process-supervision","reasoning","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"研究推理模型认知结构、轨迹标注与自动过程分析的读者。","confidence":"high","one_line":["This benchmark applies Schoenfeld’s cognitive episodes to human-annotated DeepSeek-R1 math traces at paragraph and sentence levels.","该基准以 Schoenfeld 认知情节标注 DeepSeek-R1 数学轨迹，覆盖段落和句子两个层级。"],"why":"It provides an early theory-grounded corpus and annotation guide for decomposing reasoning traces into functional steps.","primary_link":"https://arxiv.org/abs/2509.14662","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MingLiiii/Schoenfeld_Reasoning"}],"link_count":4,"sections":9},{"id":"unified-reward-model-for-multimodal-understanding-and-generation-2025","title":"Unified Reward Model for Multimodal Understanding and Generation","year":2025,"venue":"arXiv","authors":["Yibin Wang","Yuhang Zang","Hao Li","Cheng Jin","Jiaqi Wang"],"authors_zh":"Yibin Wang、Yuhang Zang、Hao Li、Cheng Jin、Jiaqi Wang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["alignment"],"tags":["preference","reward","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"偏好或奖励反馈数据论文","best_for_zh":"研究偏好学习、奖励建模或对齐的读者。","confidence":"high","one_line":["This paper releases or uses a preference or reward-feedback artifact for alignment research.","UnifiedReward 联合图像和视频的理解与生成偏好，统一支持成对排序与单点评分，并用于自动构造 DPO 数据。"],"why":"It provides a feedback object for alignment training or evaluation.","primary_link":"https://arxiv.org/abs/2503.05236","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/CodeGoat24/UnifiedReward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/CodeGoat24/unifiedreward-training-data-67c300d4fd5eff00fa7f1ede"},{"key":"project","label":["Project","项目主页"],"url":"https://codegoat24.github.io/UnifiedReward/"}],"link_count":4,"sections":9},{"id":"unitcoder-code-synthesis-unit-tests-2025","title":"UnitCoder: Scalable Code Synthesis from Pre-training Corpora","year":2025,"venue":"EMNLP 2025","authors":["Yichuan Ma","Yunfan Shao","Peiji Li","Demin Song","Qipeng Guo","Linyang Li","Xipeng Qiu","Kai Chen"],"authors_zh":"Yichuan Ma、Yunfan Shao、Peiji Li、Demin Song、Qipeng Guo、Linyang Li、Xipeng Qiu、Kai Chen","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["code-generation","unit-tests"],"tags":["code-synthesis","unit-tests","pretraining-corpora","2025"],"status":"verified","priority":"可读","paper_type_zh":"代码合成单元测试训练数据论文","best_for_zh":"研究从代码语料构建可执行训练数据或代码合成的研究者。","confidence":"high","one_line":["UnitCoder mines functions and unit tests from pre-training corpora, using test execution to filter scalable code-synthesis training data.","UnitCoder 从预训练语料中挖掘函数与单元测试，借助测试执行筛出可用于代码合成训练的大规模样本。"],"why":"UnitCoder mines functions and unit tests from pre-training corpora, using test execution to filter scalable code-synthesis training data.","primary_link":"https://aclanthology.org/2025.emnlp-main.286/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Entarochuan/UnitCoder"}],"link_count":2,"sections":9},{"id":"ursa-multimodal-math-reasoning-2025","title":"Unlocking Multimodal Mathematical Reasoning via Process Reward Model","year":2025,"venue":"NeurIPS 2025","authors":["Ruilin Luo","Zhuofan Zheng","Lei Wang","Yifan Wang","Xinzhe Ni","Zicheng Lin","Songtao Jiang","Yiyao Yu","Chufan Shi","Ruihang Chu","Jin Zeng","Yujiu Yang"],"authors_zh":"Ruilin Luo, Zhuofan Zheng, Lei Wang, Yifan Wang, Xinzhe Ni, Zicheng Lin, Songtao Jiang, Yiyao Yu, Chufan Shi, Ruihang Chu, Jin Zeng, Yujiu Yang","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["data_release","construction_recipe","verifier_reward","model_report"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward"],"training_use":["sft","process_supervision","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["multimodal_reasoning","visual_mathematics"],"tags":["multimodal-math","process-reward-model","process-supervision","visual-grounding","ps-grpo","synthetic-data"],"status":"partial","priority":"可读","paper_type_zh":"多模态过程监督数据构造、过程奖励模型与在线强化学习研究","best_for_zh":"需要审计视觉数学 CoT、过程标签、PRM 选择与在线 RL 奖励契约的读者","confidence":"medium","one_line":["URSA releases multimodal CoT and embedded-label process tables and trains a PRM for Best-of-N and PS-GRPO; the release lacks full image, trajectory, and reward-lineage records needed to audit the reported construction pipeline.","URSA 将多模态 CoT、BEL/MIE 过程标签和 PRM 分数下降信号接入 PS-GRPO；公开表格未保留完整图像、rollout 与奖励血缘。"],"why":"For rollout and search trace data, URSA separates terminal outcome, a learned process score, a detected score drop, and a modified RL reward—an essential distinction when multimodal traces are reused or audited.","primary_link":"https://papers.nips.cc/paper_files/paper/2025/hash/4757e472d97ca980c24a487622f7ff00-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/URSA-MATH/URSA-MATH"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/URSA-MATH/DualMath-1.1M"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/URSA-MATH"},{"key":"project","label":["Project","项目主页"],"url":"https://ursa-math.github.io/"}],"link_count":7,"sections":9},{"id":"posthoc-dataset-inference-2025","title":"Unlocking Post-hoc Dataset Inference with Synthetic Data","year":2025,"venue":"ICML 2025","authors":["Bihe Zhao","Pratyush Maini","Franziska Boenisch","Adam Dziedzic"],"authors_zh":"Bihe Zhao、Pratyush Maini、Franziska Boenisch、Adam Dziedzic","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","data_release","audit_failure"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["general-text","data-provenance"],"tags":["dataset-inference","synthetic-heldout","membership-inference","provenance-audit","post-hoc-calibration","false-positive-control","suffix-completion","open-release"],"status":"partial","priority":"可读","paper_type_zh":"合成参考集与数据集推断配方","best_for_zh":"研究训练数据溯源、membership inference、统计校准和审计证据边界的读者","confidence":"medium","one_line":["Shared-prefix synthetic suffixes and calibrated paired tests enable post-hoc set-level training-data inference without a pre-existing matched holdout.","该工作通过同前缀合成后缀与校准配对检验，在缺少天然匹配留出集时进行事后集合级训练数据推断，但不能给出记录级或法律层面的证明。"],"why":"The method turns a missing-heldout problem in provenance audits into an explicit data-construction and calibration recipe, while its result remains conditional aggregate evidence rather than record-level lineage or legal proof.","primary_link":"https://proceedings.mlr.press/v267/zhao25q.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sprintml/PostHocDatasetInference"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/file/d/1NijRPbnx4aSYdQuqj9jJa6LxaGBJinAA/view?usp=sharing"}],"link_count":6,"sections":9},{"id":"chain-of-step-reasoning-vlm-fine-grained-rewards-2025","title":"Unveiling Chain of Step Reasoning for Vision-Language Models with Fine-grained Rewards","year":2025,"venue":"NeurIPS 2025","authors":["Honghao Chen","Xingzhou Lou","Xiaokun Feng","Kaiqi Huang","Xinlong Wang"],"authors_zh":"Honghao Chen, Xingzhou Lou, Xiaokun Feng, Kaiqi Huang, Xinlong Wang","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","mathematical-reasoning","process-reward-modeling"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要构建或评测视觉语言模型细粒度过程奖励与推理扩展的研究者。","confidence":"high","one_line":["CoS-Dataset releases fine-grained reward supervision for vision-language reasoning steps, supporting VLM process rewards, reranking, and reinforcement learning.","CoS-Dataset 提供视觉推理步骤与细粒度奖励，可直接用于视觉语言模型的过程奖励、重排序和强化学习。"],"why":"It couples a released VLM reasoning corpus with step rewards, a PRM, and the training path that uses them.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/a68120d2eb2f53f7d9e71547591aef11-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/baaivision/CoS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Lauch1ng/CoS-Dataset"}],"link_count":4,"sections":9},{"id":"openai-gpt-5-2-system-card-2025","title":"Update to GPT-5 System Card: GPT-5.2","year":2025,"venue":"OpenAI system-card update","authors":["OpenAI"],"authors_zh":"OpenAI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode"],"training_use":["evaluation","audit","safety_alignment"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["general_reasoning","safety","cybersecurity","biosecurity","ai_self_improvement","agentic_systems","chain_of_thought"],"tags":["openai","gpt-5-2","system-card","reinforcement-learning","chain-of-thought","monitorability","prompt-injection","train-eval-overlap","production-evaluation","preparedness","agentic-evaluation","hidden-unit-tests","safety-training","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"前沿系统卡披露增量与安全审计","best_for_zh":"审计训练信号、评测 grader、Preparedness agent 环境、CoT monitor 与部署防护边界的研究者","confidence":"high","one_line":["GPT-5.2 repeats GPT-5's coarse source/reasoning-RL disclosure but adds explicit overlap boundaries, production and agent evaluation contracts, and a dated CoT-monitorability audit without releasing training records or rewards.","GPT-5.2 相对 GPT-5 没有增加训练来源或 reasoning RL 的可复现细节，但新增明确的 train/eval overlap、窄范围 non-overlap/held-out 声明、生产与 agent 评测契约及带日期的 CoT monitorability 审计。"],"why":"It prevents production samples, hidden tests, graders, CoT monitors, and deployment safeguards from being mistaken for GPT-5.2 training data.","primary_link":"https://cdn.openai.com/pdf/3a4153c8-c748-4b71-8e31-aecbde944f8d/oai_5_2_system-card.pdf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-gpt-5-2/"}],"link_count":3,"sections":9},{"id":"userbench-2025","title":"UserBench: An Interactive Gym Environment for User-Centric Agents","year":2025,"venue":"NeurIPS 2025 Workshop on Scaling Environments for Agents","authors":["Cheng Qian","Zuxin Liu","Akshara Prabhakar","Zhiwei Liu","Jianguo Zhang","Haolin Chen","Heng Ji","Weiran Yao","Shelby Heinecke","Silvio Savarese","Caiming Xiong","Huan Wang"],"authors_zh":"Cheng Qian, Zuxin Liu, Akshara Prabhakar, Zhiwei Liu, Jianguo Zhang, Haolin Chen, Heng Ji, Weiran Yao, Shelby Heinecke, Silvio Savarese, Caiming Xiong, Huan Wang","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer"],"domains":["environment_interaction","conversational_agents","tool_use","travel_planning","user_preference_elicitation"],"tags":["userbench","environment-agent-trajectory-data","agent-benchmark","conversational-agent","user-simulator","preference-elicitation","tool-use","mixed-verification","synthetic-travel","version-drift"],"status":"partial","priority":"可读","paper_type_zh":"用户偏好澄清型智能体评测环境与任务发布","best_for_zh":"研究环境智能体数据、用户模拟器、混合 verifier、偏好澄清评测与发布审计的读者","confidence":"medium","one_line":["UserBench releases 3,122 unique travel-planning tasks and a mixed GPT-4o/rule Gym environment for preference elicitation, but not the paper's rollout trajectories, and its counts, judge labels, randomness, and best-option ties require a pinned audit.","UserBench 发布 3,122 个唯一旅行规划任务及 GPT-4o/规则混合验证的 Gym 交互环境，用于评测偏好澄清、搜索与推荐；论文运行轨迹未发布，计数、judge 编号、随机性与 best_id 并列标签仍需固定版本审计。"],"why":"It shows how an interactive agent benchmark can turn clarifying questions, searches, implicit user responses, and option choices into state/action rewards, while also demonstrating why prospective train splits are not training evidence and why judge, label, split, and replay metadata determine reuse safety.","primary_link":"https://arxiv.org/abs/2507.22034","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SalesforceAIResearch/UserBench"},{"key":"data","label":["Data","数据"],"url":"https://github.com/SalesforceAIResearch/UserBench/tree/80506d2ab484cab843e60a2401ff3e0290d05b87/data"}],"link_count":11,"sections":9},{"id":"value-guided-search-cot-2025","title":"Value-Guided Search for Efficient Chain-of-Thought Reasoning","year":2025,"venue":"NeurIPS 2025","authors":["Kaiwen Wang","Jin Peng Zhou","Jonathan Chang","Zhaolin Gao","Nathan Kallus","Kianté Brantley","Wen Sun"],"authors_zh":"Kaiwen Wang, Jin Peng Zhou, Jonathan Chang, Zhaolin Gao, Nathan Kallus, Kianté Brantley, Wen Sun","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["scalar_reward","trajectory_value"],"training_use":["reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer"],"domains":["mathematics","competition-math","reasoning"],"tags":["value-guided-search","openr1-vm","token-level-value-model","roll-in-roll-out","block-wise-beam-search","dvts","weighted-majority-vote","test-time-compute","math-verify","negative-traces"],"status":"partial","priority":"必读","paper_type_zh":"长链推理价值模型数据发布与 test-time search 配方","best_for_zh":"研究 rollout/search trace、value model 监督、推理预算归因和开放数据审计的读者","confidence":"high","one_line":["VGS releases 2.5M math roll-in/roll-out pairs with correct, incorrect, and incomplete outcome labels, trains a 1.5B token-level value model, and uses it to guide 4,096-token beam-search blocks plus weighted voting.","VGS 公开 250 万条带正确、错误与未完成结果标签的数学 roll-in/roll-out 对，并用其训练 1.5B token-level value model 引导分块搜索；完整拒绝清单、历史搜索树及代码和模型许可证仍未公开。"],"why":"It connects the full reasoning-data lifecycle—prompt filtering, failed and successful rollouts, token-level value targets, selector-guided search, and explicit compute budgets—while making the missing rejection manifests, discarded branches, and license gaps visible for audit.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/7a8a3a34ede2bae1fd2fb1876a7ba362-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kaiwenw/ValueGuidedSearch"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/VGS-AI/OpenR1-VM"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/VGS-AI"}],"link_count":7,"sections":9},{"id":"vapr-vision-language-preference-alignment-reasoning-2025","title":"VaPR: Vision-language Preference Alignment for Reasoning","year":2025,"venue":"COLM 2025","authors":["Rohan Wadhawan","Fabrice Y. Harel-Canada","Zi-Yi Dou","Suhaila Shakiah","Robinson Piramuthu","Nanyun Peng"],"authors_zh":"Rohan Wadhawan、Fabrice Y. Harel-Canada、Zi-Yi Dou、Suhaila Shakiah、Robinson Piramuthu、Nanyun Peng","tracks":["preference_reward_feedback_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","vision-language"],"tags":["vision-language","dpo","preference-data","colm-2025"],"status":"verified","priority":"可读","paper_type_zh":"视觉语言偏好数据集与对齐方法","best_for_zh":"需要多模态 DPO、视觉推理偏好对或困难负例构造的研究者","confidence":"high","one_line":["VaPR releases 30,000 vision-language preference pairs whose edited hard negatives control for style and length while targeting reasoning errors.","VaPR-30K 发布超过 3 万视觉推理 chosen/rejected 对，并以困难负例强化细粒度视觉—语言推理偏好学习。"],"why":"Its hard negatives are designed to change reasoning quality without giving away the label through superficial response cues.","primary_link":"https://openreview.net/forum?id=uBAubFwymy","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/VaPR-UCLA/vapr-30k"},{"key":"project","label":["Project","项目主页"],"url":"https://vap-r.github.io/"}],"link_count":3,"sections":9},{"id":"vericoder-functional-correctness-validation-2025","title":"VeriCoder: Enhancing LLM-Based RTL Code Generation through Functional Correctness Validation","year":2025,"venue":"arXiv","authors":["Anjiang Wei","Huanmi Tan","Tarun Suresh","Daniel Mendoza","Thiago S. F. X. Teixeira","Ke Wang","Caroline Trippel","Alex Aiken"],"authors_zh":"Anjiang Wei, Huanmi Tan, Tarun Suresh, Daniel Mendoza, Thiago S. F. X. Teixeira, Ke Wang, Caroline Trippel, Alex Aiken","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","reward_modeling","rlvr","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["code-generation","hardware-design"],"tags":["rtl","verilog","hardware-design","simulation","executable-verification",2025],"status":"verified","priority":"可读","paper_type_zh":"功能验证的 RTL 代码数据与生成模型","best_for_zh":"研究 RTL 代码生成、硬件验证或执行式奖励的研究者。","confidence":"high","one_line":["VeriCoder turns natural-language RTL seeds into 125,777 simulation-passing specification–RTL–test triplets through teacher-guided test-and-repair loops.","VeriCoder 通过教师生成测试、仿真反馈和迭代修复，把 RTL 种子扩展为 125,777 条功能验证的规格—实现—测试三元组。"],"why":"It changes RTL data acceptance from compilation to behavior under an executable testbench while preserving the validation evidence in each record.","primary_link":"https://arxiv.org/abs/2504.15659","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Anjiang-Wei/VeriCoder"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LLM4Code/expanded_origen_126k"},{"key":"project","label":["Project","项目主页"],"url":"https://anjiang-wei.github.io/VeriCoder-Website/"}],"link_count":5,"sections":9},{"id":"vericontaminated-2025","title":"VeriContaminated: Assessing LLM-Driven Verilog Coding for Data Contamination","year":2025,"venue":"ICLAD 2025","authors":["Zeng Wang","Minghao Shao","Jitendra Bhandari","Likhitha Mankali","Ramesh Karri","Ozgur Sinanoglu","Muhammad Shafique","Johann Knechtel"],"authors_zh":"Zeng Wang，Minghao Shao，Jitendra Bhandari，Likhitha Mankali，Ramesh Karri，Ozgur Sinanoglu，Muhammad Shafique，Johann Knechtel。","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["track13","audit","final-slate"],"status":"verified","priority":"必读","paper_type_zh":"Verilog 代码生成基准的数据污染与评测可靠性审计","best_for_zh":"评估硬件代码生成模型、构建去污染基准或审计训练—测试泄漏的研究者。","confidence":"medium","one_line":["Official repository releases the Verilog contamination-assessment implementation.","以双检测器审计 Verilog 评测污染，并量化缓解污染与代码质量的取舍。"],"why":"It adds an auditable reliability or failure-mode surface to Track 13.","primary_link":"https://doi.org/10.1109/ICLAD65226.2025.00017","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DfX-NYUAD/VeriContaminated"}],"link_count":3,"sections":9},{"id":"verif-instruction-following-2025","title":"VerIF: Verification Engineering for Reinforcement Learning in Instruction Following","year":2025,"venue":"EMNLP 2025","authors":["Hao Peng","Yunjia Qi","Xiaozhi Wang","Bin Xu","Lei Hou","Juanzi Li"],"authors_zh":"Hao Peng、Yunjia Qi、Xiaozhi Wang、Bin Xu、Lei Hou、Juanzi Li","tracks":["training_usage_optimization_objectives"],"source_role":["verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["scalar_reward"],"training_use":["rlvr"],"construction_layer":["reward_verifier_layer"],"domains":["instruction-tuning"],"tags":["post-training","training-usage"],"status":"verified","priority":"必读","paper_type_zh":"指令遵循强化学习验证器与可验证数据论文","best_for_zh":"需要为复杂自然语言指令构造可审计强化学习奖励的读者。","confidence":"high","one_line":["VerIF engineers hard and soft instruction verifiers into reward signals for reinforcement learning on instruction following.","VerIF 将硬约束和软约束指令验证器工程化为奖励信号，用于指令遵循的强化学习。"],"why":"It makes the connection between a data object and a training objective inspectable.","primary_link":"https://aclanthology.org/2025.emnlp-main.1542/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THU-KEG/VerIF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THU-KEG/VerInstruct"}],"link_count":3,"sections":9},{"id":"verifact-long-form-factuality-2025","title":"VeriFact: Enhancing Long-Form Factuality Evaluation with Refined Fact Extraction and Reference Facts","year":2025,"venue":"EMNLP 2025","authors":["Xin Liu","Lechen Zhang","Sheza Munir","Yiyang Gu","Lu Wang"],"authors_zh":"Xin Liu、Lechen Zhang、Sheza Munir、Yiyang Gu、Lu Wang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["VeriFact refines fact extraction and introduces FactRBench to evaluate both precision and recall of long-form factuality.","VeriFact 用精炼事实抽取和参考事实集 FactRBench，同时评估长回答事实的精确率与召回率。"],"why":"VeriFact refines fact extraction and introduces FactRBench to evaluate both precision and recall of long-form factuality.","primary_link":"https://aclanthology.org/2025.emnlp-main.905/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/launch/FactRBench"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/launch/factrbench"}],"link_count":3,"sections":9},{"id":"verifiable-format-control-vff-2025","title":"Verifiable Format Control for Large Language Model Generations","year":2025,"venue":"Findings of NAACL 2025","authors":["Zhaoyang Wang","Jinqi Jiang","Huichi Zhou","Wenhao Zheng","Xuchao Zhang","Chetan Bansal","Huaxiu Yao"],"authors_zh":"Zhaoyang Wang、Jinqi Jiang、Huichi Zhou、Wenhao Zheng、Xuchao Zhang、Chetan Bansal、Huaxiu Yao","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["structured-generation","format-control"],"tags":["format-control","structured-output","verification","2025"],"status":"verified","priority":"可读","paper_type_zh":"可验证格式控制数据集与评测论文","best_for_zh":"需要把结构化输出约束变成自动验证信号的研究者。","confidence":"high","one_line":["VFF builds automatically checkable format-control tasks that train and evaluate strict adherence to output constraints.","VFF 构造可自动判定的格式控制任务，以严格语法约束评估并训练模型遵循输出格式。"],"why":"VFF builds automatically checkable format-control tasks that train and evaluate strict adherence to output constraints.","primary_link":"https://aclanthology.org/2025.findings-naacl.194/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/jinqij/VFF"}],"link_count":2,"sections":9},{"id":"verifierq-qlearning-verifiers-2025","title":"VerifierQ: Enhancing LLM Test Time Compute with Q-Learning-based Verifiers","year":2025,"venue":"AI4Math Workshop at ICML 2025 Poster","authors":["Jianing Qi","Hao Tang","Zhigang Zhu"],"authors_zh":"Jianing Qi（CUNY 研究生中心）；Hao Tang（CUNY 研究生中心、曼哈顿社区学院）；Zhigang Zhu（CUNY 研究生中心、纽约市立学院）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["process_supervision","test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","verifier","q-learning","process-reward-model","mathematical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"面向测试时扩展的离线 Q-learning 验证器研究","best_for_zh":"研究过程验证器、数学推理候选选择与验证器标度的读者。","confidence":"high","one_line":["VerifierQ trains an utterance-level Q-value verifier offline, aiming to select stronger multistep mathematical solutions under a test-time compute budget.","VerifierQ 用离线 Q-learning 训练步骤级验证器，在增加推理候选时更有效地评估和选择数学解答。"],"why":"It exposes verifier training, not just generator sampling, as a scaling lever and tests it against PRM and majority voting.","primary_link":"https://arxiv.org/abs/2410.08048","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/meta-math/MetaMathQA"}],"link_count":4,"sections":9},{"id":"verithoughts-verilog-formal-verification-2025","title":"VeriThoughts: Enabling Automated Verilog Code Generation using Reasoning and Formal Verification","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks","authors":["Yubo Wang","Yash Sharma","Hao Luo","Yingcong Chen","Prithviraj Sen","Sanjay Shakkottai","Swarup Bhunia"],"authors_zh":"Yubo Wang、Yash Sharma、Hao Luo、Yingcong Chen、Prithviraj Sen、Sanjay Shakkottai、Swarup Bhunia","tracks":["programmatically_verifiable_outcome_data"],"source_role":["data_release"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","rlvr"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["verilog","hardware-design","code-generation"],"tags":["verilog","formal-verification","hardware","2025"],"status":"verified","priority":"可读","paper_type_zh":"Verilog 推理与形式验证训练数据集","best_for_zh":"研究硬件代码生成、形式验证反馈或 RTL 对齐的研究者。","confidence":"high","one_line":["VeriThoughts filters Verilog reasoning-and-code traces with formal-verification feedback for synthesizable RTL generation.","VeriThoughts 用形式验证反馈筛选 Verilog 推理与代码轨迹，面向可综合 RTL 生成。"],"why":"VeriThoughts filters Verilog reasoning-and-code traces with formal-verification feedback for synthesizable RTL generation.","primary_link":"https://arxiv.org/abs/2505.20302","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wilyub/VeriThoughts"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/wilyub/VeriThoughtsTrainSet"}],"link_count":3,"sections":9},{"id":"verltool-holistic-agentic-rl-with-tool-use-2025","title":"VerlTool: Towards Holistic Agentic Reinforcement Learning with Tool Use","year":2025,"venue":"Transactions on Machine Learning Research (TMLR), 2026","authors":["Dongfu Jiang","Yi Lu","Zhuofeng Li","Zhiheng Lyu","Ping Nie","Haozhe Wang","Alex Su","Hui Chen","Kai Zou","Chao Du","Tianyu Pang","Wenhu Chen"],"authors_zh":"Dongfu Jiang, Yi Lu, Zhuofeng Li, Zhiheng Lyu, Ping Nie, Haozhe Wang, Alex Su, Hui Chen, Kai Zou, Chao Du, Tianyu Pang, Wenhu Chen","tracks":["environment_agent_trajectory_data"],"source_role":["infrastructure","agent_environment","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["rlvr","agent_training","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["agent_trajectories","environment_interaction","tool_use","mathematics","information_retrieval","text_to_sql","multimodal_reasoning","deep_research","software_engineering"],"tags":["environment-agent-trajectory-data","agentic-rl","tool-use","multi-turn-trajectories","environment-feedback","rlvr","verl"],"status":"partial","priority":"可读","paper_type_zh":"智能体强化学习基础设施与环境构造配方","best_for_zh":"研究工具使用智能体、环境反馈、RLVR 训练基础设施和轨迹可复现审计的读者","confidence":"medium","one_line":["VerlTool turns task prompts into on-policy, multi-turn action-observation episodes behind a stateful Tool Server, but releases no paper-pinned six-domain trajectory corpus or replay manifest.","VerlTool 以状态化 Tool Server 将六类异构任务转换为在线多轮 action-observation episode，并用领域标量奖励训练智能体；其价值在统一构造与反馈接口，但未发布论文版本固定的六域轨迹语料或 replay manifest。"],"why":"Its unified interface makes environment observations, validity/termination signals and terminal reward provenance comparable across math, search, SQL, visual, deep-search and SWE agents. The absent episode release and post-paper code drift sharply limit direct data reuse and exact replay.","primary_link":"https://openreview.net/pdf?id=g2LCOW43Md","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TIGER-AI-Lab/verl-tool"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/VerlTool"}],"link_count":8,"sections":9},{"id":"versaprm-multidomain-process-reward-2025","title":"VersaPRM: Multi-Domain Process Reward Model via Synthetic Reasoning Data","year":2025,"venue":"ICML 2025 Oral","authors":["Thomas Zeng","Shuibai Zhang","Shutong Wu","Christian Classen","Daewon Chae","Ethan Ewer","Minjae Lee","Heeju Kim","Wonjun Kang","Jackson Kunde","Ying Fan","Jungtaek Kim","Hyung Il Koo","Kannan Ramchandran","Dimitris Papailiopoulos","Kangwook Lee"],"authors_zh":"Thomas Zeng、Shuibai Zhang、Shutong Wu、Christian Classen 等（威斯康星大学麦迪逊分校、高丽大学、FuriosaAI、加州大学伯克利分校）","tracks":["training_usage_optimization_objectives","process_trace_supervision_data"],"source_role":["process_supervision","construction_recipe","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["process_supervision","test_time_compute"],"construction_layer":["reward_verifier_layer"],"domains":["mathematical-reasoning","science","law","philosophy","multi-domain-reasoning"],"tags":["process-reward-model","synthetic-reasoning-data","step-labels","verifier","multi-domain"],"status":"verified","priority":"必读","paper_type_zh":"合成过程监督数据发布与多领域过程奖励模型研究","best_for_zh":"适合构建过程标签、奖励模型或非狭义数学任务测试时搜索系统的读者。","confidence":"high","one_line":["VersaPRM creates and releases multi-domain step-labeled reasoning traces, then trains a process reward model that improves test-time selection beyond mathematics.","VersaPRM 构造并发布多领域、带步骤标签的推理轨迹，再训练过程奖励模型，使测试时选择能力扩展到数学以外的领域。"],"why":"It demonstrates that diversity and label construction in process data determine where a reward model remains useful, not merely which optimizer is applied.","primary_link":"https://proceedings.mlr.press/v267/zeng25h.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UW-Madison-Lee-Lab/VersaPRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/UW-Madison-Lee-Lab/MMLU-Pro-CoT-Train-Labeled"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/UW-Madison-Lee-Lab/VersaPRM"}],"link_count":6,"sections":9},{"id":"vibe-code-bench-2025","title":"Vibe Code Bench v1.1: Evaluating AI Models on End-to-End Web Application Development","year":2025,"venue":"Vals AI benchmark page","authors":["Vals AI"],"authors_zh":"Vals AI","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["coding","web-generation","product-prototyping"],"tags":["benchmark","coding","evaluation-surface","web-generation"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"关注端到端代码生成、网页应用实现、功能测试和工程交付质量的读者。","confidence":"high","one_line":["Vibe Code Bench evaluates whether models can build functioning web applications in a full development environment, scored through point-and-click functional tests and review criteria.","Vibe Code Bench 评测模型在完整开发环境中生成可用网页应用的能力，核心反馈来自交互式功能测试和工程审核准则。"],"why":"Vibe Code Bench evaluates whether models can build functioning web applications in a full development environment, scored through point-and-click functional tests and review criteria.","primary_link":"https://www.vals.ai/benchmarks/vibe-code","links":[],"link_count":3,"sections":9},{"id":"videorewardbench-video-understanding-2025","title":"VideoRewardBench: Comprehensive Evaluation of Multimodal Reward Models for Video Understanding","year":2025,"venue":"arXiv 2025","authors":["Zhihong Zhang","Xiaojian Huang","Jin Xu","Zhuodong Luo","Xinzhi Wang","Jiansheng Wei","Xuejin Chen"],"authors_zh":"Zhihong Zhang, Xiaojian Huang, Jin Xu, Zhuodong Luo, Xinzhi Wang, Jiansheng Wei, Xuejin Chen","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["video-understanding","multimodal-reward","preference-evaluation"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"视频理解多模态奖励模型的偏好评测基准论文","best_for_zh":"需要检验视频奖励模型能否在感知、知识、推理与安全间稳定选择优质回答的研究者。","confidence":"high","one_line":["VideoRewardBench exposes whether a reward model can rank answers to real videos across perception, knowledge, reasoning, and safety.","1,563 个视频偏好样本、1,482 段视频，按感知、知识、推理、安全给出 chosen/rejected 回答，适合视频推理反馈。"],"why":"It extends multimodal preference feedback from still images to temporal understanding.","primary_link":"https://arxiv.org/abs/2509.00484","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zhang123434/videorewardbench"},{"key":"project","label":["Project","项目主页"],"url":"https://videorewardbench.github.io/"}],"link_count":4,"sections":9},{"id":"vilbench-2025","title":"ViLBench: A Suite for Vision-Language Process Reward Modeling","year":2025,"venue":"EMNLP 2025","authors":["Haoqin Tu","Weitao Feng","Hardy Chen","Hui Liu","Xianfeng Tang","Cihang Xie"],"authors_zh":"Haoqin Tu、Weitao Feng、Hardy Chen、Hui Liu、Xianfeng Tang、Cihang Xie","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["data_release","benchmark","verifier_reward","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward","trajectory_value"],"training_use":["reward_modeling","test_time_compute","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","search_substrate","release_audit"],"domains":["multimodal_reasoning","visual_reasoning","mathematics","science"],"tags":["multimodal-reasoning","process-reward-model","mcts","test-time-selection","vision-language"],"status":"partial","priority":"可读","paper_type_zh":"多模态过程奖励建模、MCTS 轨迹构造与测试时选择研究","best_for_zh":"需要审计多模态推理轨迹、终局裁判如何传导为步骤价值，以及 PRM 用于 Best-of-N 选择边界的读者","confidence":"high","one_line":["A multimodal PRM benchmark and a released 73,560-row MCTS-derived process/value dataset, where terminal GPT-4o judgments are propagated into continuous partial-solution scores.","ViLBench 同时提供面向过程奖励的多模态评测集与公开的 ViLReward-73K：后者把图像条件下 MCTS 的部分推理过程及其连续价值写成 73,560 行数据。"],"why":"","primary_link":"https://aclanthology.org/2025.emnlp-main.344/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/UCSC-VLAA/ViLBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/UCSC-VLAA/ViLReward-73K"},{"key":"project","label":["Project","项目主页"],"url":"https://ucsc-vlaa.github.io/ViLBench/"}],"link_count":7,"sections":9},{"id":"visco-visual-reasoning-critique-2025","title":"VISCO: Benchmarking Fine-Grained Critique and Correction Towards Self-Improvement in Visual Reasoning","year":2025,"venue":"CVPR 2025","authors":["Xueqing Wu","Yuheng Ding","Bingxuan Li","Pan Lu","Da Yin","Kai-Wei Chang","Nanyun Peng"],"authors_zh":"Xueqing Wu、Yuheng Ding、Bingxuan Li、Pan Lu、Da Yin、Kai-Wei Chang、Nanyun Peng","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["VISCO measures fine-grained critique and critique-conditioned correction for visual reasoning.","针对视觉推理答案同时评估细粒度批评与基于批评的修正能力。"],"why":"VISCO measures fine-grained critique and critique-conditioned correction for visual reasoning.","primary_link":"https://openaccess.thecvf.com/content/CVPR2025/html/Wu_VISCO_Benchmarking_Fine-Grained_Critique_and_Correction_Towards_Self-Improvement_in_Visual_Reasoning_CVPR_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PlusLabNLP/VISCO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/uclanlp/VISCO"},{"key":"project","label":["Project","项目主页"],"url":"https://visco-benchmark.github.io/"}],"link_count":4,"sections":9},{"id":"vlm-self-improve-reflection-2025","title":"Vision-Language Models Can Self-Improve Reasoning via Reflection","year":2025,"venue":"NAACL 2025","authors":["Kanzhi Cheng","Yantao Li","Fangzhi Xu","Jianbing Zhang","Hao Zhou","Yang Liu"],"authors_zh":"Kanzhi Cheng、Yantao Li、Fangzhi Xu、Jianbing Zhang、Hao Zhou、Yang Liu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode"],"training_use":["sft","test_time_compute"],"construction_layer":["trace_writing","search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold"],"domains":["multimodal-reasoning","visual-mathematics","chart-reasoning","geometry","visual-question-answering","web-navigation","science-reasoning","commonsense-reasoning"],"tags":["r3v","multimodal-reasoning","iterative-self-training","positive-negative-rollouts","self-reflection","rationale-refinement","candidate-selection","test-time-selection","answer-level-verification","noisy-cot","missing-derived-data"],"status":"partial","priority":"必读","paper_type_zh":"多模态迭代自训练与测试时选择配方","best_for_zh":"关注多模态 rollout、自训练数据构造、候选选择、结果验证与 rationale 忠实性审计的研究者","confidence":"high","one_line":["R3V repeatedly samples answer-labelled multimodal CoT, trains on a latest positive, a negative-to-positive refinement, and three-candidate answer selection, then reuses the selector for test-time compute.","R3V 反复采样按最终答案标注的多模态 CoT，以最新正例、负例条件下的正例模仿和三候选答案选择进行训练，并在测试时复用选择能力。"],"why":"It connects iterative rollout refresh, negative-context SFT and candidate-conditioned inference in one inspectable recipe, while showing directly that gold-answer correctness can retain unfaithful multimodal rationales and that missing frozen rollout/decision ledgers limit reuse.","primary_link":"https://aclanthology.org/2025.naacl-long.447/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/njucckevin/MM-Self-Improve"},{"key":"data","label":["Data","数据"],"url":"https://github.com/njucckevin/MM-Self-Improve/tree/master/data/data_self_train"}],"link_count":6,"sections":9},{"id":"visualprm-2025","title":"VisualPRM: An Effective Process Reward Model for Multimodal Reasoning","year":2025,"venue":"arXiv preprint","authors":["Weiyun Wang","Zhangwei Gao","Lianjie Chen","Zhe Chen","Jinguo Zhu","Xiangyu Zhao","Yangzhou Liu","Yue Cao","Shenglong Ye","Xizhou Zhu","Lewei Lu","Haodong Duan","Yu Qiao","Jifeng Dai","Wenhai Wang"],"authors_zh":"Weiyun Wang、Zhangwei Gao、Lianjie Chen、Zhe Chen、Jinguo Zhu、Xiangyu Zhao、Yangzhou Liu、Yue Cao、Shenglong Ye、Xizhou Zhu、Lewei Lu、Haodong Duan、Yu Qiao、Jifeng Dai、Wenhai Wang","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","benchmark","construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["step_level","scalar_reward","process_reward","trajectory_value"],"training_use":["process_supervision","reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","visual-mathematics","visual-question-answering","science","charts","ocr","documents"],"tags":["visualprm","multimodal-process-reward","monte-carlo-labeling","best-of-n","visualprocessbench","open-release","rollout-audit"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["VisualPRM400K converts 16-continuation prefix success rates into binary multimodal step labels for an 8B PRM, but the original release omits the continuation and decision ledger behind those labels.","VisualPRM400K 把 16 次续写的前缀成功率转成二元多模态步骤标签用于训练 8B 过程奖励模型，但原始发布未包含这些标签背后的续写记录与判定台账。"],"why":"It is a concrete open example of how training-time rollout labels feed a process scorer and then a test-time Best-of-N selector, making rollout budget, terminal checking, version identity, and rejected traces central audit objects rather than implementation trivia.","primary_link":"https://arxiv.org/abs/2503.10291","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenGVLab/InternVL/tree/main/internvl_chat/tools/reasoning_data_pipeline"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenGVLab/VisualPRM400K"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/OpenGVLab/VisualPRM-8B"},{"key":"project","label":["Project","项目主页"],"url":"https://internvl.github.io/blog/2025-03-13-VisualPRM/"}],"link_count":6,"sections":9},{"id":"visualwebinstruct-2025","title":"VisualWebInstruct: Scaling up Multimodal Instruction Data through Web Search","year":2025,"venue":"EMNLP 2025 Main","authors":["Yiming Jia","Jiachen Li","Xiang Yue","Bo Li","Ping Nie","Kai Zou","Wenhu Chen"],"authors_zh":"Yiming Jia, Jiachen Li, Xiang Yue, Bo Li, Ping Nie, Kai Zou, Wenhu Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","mathematics","science"],"tags":["multimodal-instruction-data","visual-reasoning","web-mining","synthetic-solutions"],"status":"verified","priority":"可读","paper_type_zh":"多模态推理指令数据集与构造流程","best_for_zh":"适合构建或审计网页来源多模态 SFT 语料的研究者。","confidence":"high","one_line":["VisualWebInstruct turns image-guided web search into 906,160 multimodal and text QA demonstrations through extraction, consistency filtering, and source-answer alignment.","VisualWebInstruct 通过图像引导的网页搜索、答案一致性筛选与网页答案对齐，构造了 906,160 条多模态和纯文本问答示范。"],"why":"It provides a concrete large-scale recipe for obtaining visual reasoning supervision beyond narrow synthetic diagrams or small academic datasets.","primary_link":"https://arxiv.org/abs/2503.10582","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TIGER-AI-Lab/VisualWebInstruct"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TIGER-Lab/VisualWebInstruct"},{"key":"project","label":["Project","项目主页"],"url":"https://tiger-ai-lab.github.io/VisualWebInstruct/"}],"link_count":7,"sections":9},{"id":"vl-rewardbench-vision-language-generative-reward-models-2025","title":"VL-RewardBench: A Challenging Benchmark for Vision-Language Generative Reward Models","year":2025,"venue":"CVPR 2025","authors":["Lei Li","Yuancheng Wei","Zhihui Xie","Xuqing Yang","Yifan Song","Peiyi Wang","Chenxin An","Tianyu Liu","Sujian Li","Bill Yuchen Lin","Lingpeng Kong","Qi Liu"],"authors_zh":"Lei Li、Yuancheng Wei、Zhihui Xie、Xuqing Yang、Yifan Song、Peiyi Wang、Chenxin An、Tianyu Liu、Sujian Li、Bill Yuchen Lin、Lingpeng Kong、Qi Liu","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","evaluation","test_time_compute"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["multimodal-reasoning","vision-language","reward-modeling"],"tags":["vision-language","reward-benchmark","human-verification","cvpr-2025"],"status":"verified","priority":"可读","paper_type_zh":"视觉语言生成奖励模型基准","best_for_zh":"需要评测多模态奖励模型、judge 或候选答案重排器的研究者","confidence":"high","one_line":["VL-RewardBench is a 1,250-example human-verified benchmark that stress-tests vision-language generative reward models on perception, hallucination, and reasoning.","1,250 个经人工核验的图文偏好样本覆盖感知、幻觉与数学推理，是多模态生成奖励模型的核心评测数据。"],"why":"It separates basic visual-perception failures, hallucinations, and reasoning failures in a public preference benchmark.","primary_link":"https://openaccess.thecvf.com/content/CVPR2025/html/Li_VL-RewardBench_A_Challenging_Benchmark_for_Vision-Language_Generative_Reward_Models_CVPR_2025_paper.html","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Zhihui/Vl-RewardBench"},{"key":"project","label":["Project","项目主页"],"url":"https://vl-rewardbench.github.io/"}],"link_count":4,"sections":9},{"id":"vlrmbench-vision-language-reward-models-2025","title":"VLRMBench: A Comprehensive and Challenging Benchmark for Vision-Language Reward Models","year":2025,"venue":"ICCV 2025","authors":["Jiacheng Ruan","Wenzhen Yuan","Xian Gao","Ye Guo","Daoxin Zhang","Zhe Xu","Yao Hu","Ting Liu","Yuzhuo Fu"],"authors_zh":"Jiacheng Ruan, Wenzhen Yuan, Xian Gao, Ye Guo, Daoxin Zhang, Zhe Xu, Yao Hu, Ting Liu, Yuzhuo Fu","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multimodal-reasoning","reward-modeling","evaluation"],"tags":["process-supervision","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要全面评测视觉语言奖励模型与过程判别能力的研究者。","confidence":"high","one_line":["VLRMBench provides 12,634 vision-language reward-model questions spanning process understanding, outcome judgment, and critique.","VLRMBench 含 12,634 个视觉语言问题，覆盖过程理解、结果判断与批评生成，可系统评测视觉奖励模型。"],"why":"The release provides a reusable process-supervision or process-evaluation surface.","primary_link":"https://openaccess.thecvf.com/content/ICCV2025/html/Ruan_VLRMBench_A_Comprehensive_and_Challenging_Benchmark_for_Vision-Language_Reward_Models_ICCV_2025_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/JCruan519/VLRMBench"},{"key":"data","label":["Data","数据"],"url":"https://github.com/JCruan519/VLRMBench/tree/main/benchmark_data"}],"link_count":4,"sections":9},{"id":"wasp-web-agent-security-2025","title":"WASP: Benchmarking Web Agent Security Against Prompt Injection Attacks","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Ivan Evtimov","Arman Zharmagambetov","Aaron Grattafiori","Chuan Guo","Kamalika Chaudhuri"],"authors_zh":"Ivan Evtimov, Arman Zharmagambetov, Aaron Grattafiori, Chuan Guo, Kamalika Chaudhuri","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","audit_failure"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["full_episode"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","web_navigation","web_agent_security","prompt_injection"],"tags":["environment-agent-trajectory-data","benchmarks-evaluation-surfaces","web-agents","prompt-injection","agent-security","visualwebarena","mixed-verification","benchmark-audit"],"status":"partial","priority":"可读","paper_type_zh":"Web agent prompt injection安全基准与评测契约审计","best_for_zh":"研究web agent安全、环境轨迹评测、LLM-as-judge与verifier失效模式的读者","confidence":"high","one_line":["WASP operationalizes realistic web-agent hijacking as 84 runtime-instantiated tasks with a GPT-4o any-compromised-action judge and task-specific environment or action-log predicates, exposing both security-by-incompetence and verifier-contract drift.","WASP以84个运行时生成的GitLab/Postmill prompt-injection任务审计web agent，并分离GPT-4o逐动作劫持判定与任务特定的端到端成功，但未发布canonical rollout且固定commit的exfil evaluator只检查action text中的URL片段。"],"why":"It shows why an agent trajectory needs two distinct audit layers: a judge can detect that the policy started following an injection, while final-state checks reveal whether harm was actually completed; the released exfiltration checker, missing rollouts, mixed license, and environment drift also demonstrate why benchmark code is not automatically a reusable training corpus.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/1c9818387f5dd0a0bc151214660f059d-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/facebookresearch/wasp"},{"key":"data","label":["Data","数据"],"url":"https://github.com/facebookresearch/wasp/tree/ffee6f41fde76acd14bd792db442479c506260c2/webarena_prompt_injections/configs/croissant"}],"link_count":8,"sections":9},{"id":"weaver-weak-verifiers-2025","title":"Weaver: Shrinking the Generation-Verification Gap by Scaling Compute for Verification","year":2025,"venue":"NeurIPS 2025","authors":["Jon Saad-Falcon","E. Kelly Buchanan","Mayee F. Chen","Tzu-Heng Huang","Brendan McLaughlin","Tanvir Bhathal","Shang Zhu","Ben Athiwaratkun","Frederic Sala","Scott Linderman","Azalia Mirhoseini","Christopher Ré"],"authors_zh":"Jon Saad-Falcon, E. Kelly Buchanan, Mayee F. Chen等","tracks":["rollout_search_test_time_trace_data","audit_failure_contamination_verifier_attacks","scaling_rlvr_test_time_compute"],"source_role":["verifier_reward","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["test_time_compute","audit"],"construction_layer":["reward_verifier_layer","scaling_report"],"domains":["reasoning"],"tags":["track5","verifier_selection_audit"],"status":"partial","priority":"必读","paper_type_zh":"Rollout、搜索或测试时推理轨迹研究","best_for_zh":"需要审计测试时轨迹、反馈与选择机制的读者","confidence":"high","one_line":["Weaver converts reward-model and LM-judge outputs over repeated candidates into posterior correctness scores for response selection, then distills those ensemble scores into compact verifiers.","Weaver 汇聚多个弱验证器的输出，用估计准确度的归一化分数筛选重复采样候选。"],"why":"It makes verifier aggregation a separate test-time scaling axis and identifies the score, threshold, prior, filtering, candidate-pool, and lineage records needed to audit selected reasoning traces.","primary_link":"https://arxiv.org/abs/2506.18203","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HazyResearch/scaling-verification"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/hazyresearch/weaver"},{"key":"project","label":["Project","项目主页"],"url":"https://hazyresearch.stanford.edu/blog/2025-06-18-weaver"}],"link_count":7,"sections":9},{"id":"web-shepherd-advancing-prms-reinforcing-web-agents-2025","title":"Web-Shepherd: Advancing PRMs for Reinforcing Web Agents","year":2025,"venue":"NeurIPS 2025 (Spotlight)","authors":["Hyungjoo Chae","Sunghwan Kim","Junhee Cho","Seungone Kim","Seungjun Moon","Gyeom Hwangbo","Dongha Lim","Minjin Kim","Yeonjun Hwang","Minju Gwak","Dongwook Choi","Minseok Kang","Gwanhoon Im","ByeongUng Cho","Hyojun Kim","Jun Hee Han","Taeyoon Kwon","Minju Kim","Beong-woo Kwak","Dongjin Kang","Jinyoung Yeo"],"authors_zh":"Hyungjoo Chae、Sunghwan Kim、Junhee Cho 等","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["web_agents","process_supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"必读","paper_type_zh":"网页智能体过程奖励模型与步骤级偏好监督数据论文","best_for_zh":"训练或评测网页智能体的过程奖励、步骤选择与轨迹重排序。","confidence":"high","one_line":["Web-Shepherd releases step-level web-agent preference pairs and checklists to train process reward models that guide trajectory selection.","Web-Shepherd 发布网页智能体的步骤级偏好对与检查清单，用于训练引导轨迹选择的过程奖励模型。"],"why":"It provides a concrete, reusable process-level supervision object.","primary_link":"https://arxiv.org/abs/2505.15277","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kyle8581/Web-Shepherd"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/LangAGI-Lab/WebPRMCollection_preference_pair"}],"link_count":3,"sections":9},{"id":"webagent-r1-2025","title":"WebAgent-R1: Training Web Agents via End-to-End Multi-Turn Reinforcement Learning","year":2025,"venue":"EMNLP 2025","authors":["Zhepei Wei","Wenlin Yao","Yao Liu","Weizhi Zhang","Qin Lu","Liang Qiu","Changlong Yu","Puyang Xu","Chao Zhang","Bing Yin","Hyokun Yun","Lihong Li"],"authors_zh":"Zhepei Wei、Wenlin Yao、Yao Liu、Weizhi Zhang、Qin Lu、Liang Qiu、Changlong Yu、Puyang Xu、Chao Zhang、Bing Yin、Hyokun Yun、Lihong Li","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["web","agents"],"tags":["webagent-r1","webarena-lite","multi-turn-rl","m-grpo","online-rollouts","rule-based-reward","test-time-scaling"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"high","one_line":["WebAgent-R1 combines a 9,460-trajectory behavior-cloning warm-up with asynchronous current-policy WebArena rollouts, M-GRPO, and binary terminal rewards, while the online trajectories and checkpoints remain unreleased.","WebAgent-R1 用 9,460 条轨迹做行为克隆预热，配合异步的当前策略 WebArena 采样、M-GRPO 与二元终止奖励；在线轨迹与检查点均未发布。"],"why":"It makes grouped browser episodes and interaction budgets explicit post-training objects, allowing the atlas to separate BC data, online rollout generation, environmental verification, and test-time scaling.","primary_link":"https://aclanthology.org/2025.emnlp-main.401/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/weizhepei/WebAgent-R1"}],"link_count":5,"sections":9},{"id":"webchorearena-2025","title":"WebChoreArena: Evaluating Web Browsing Agents on Realistic Tedious Web Tasks","year":2025,"venue":"COLM 2026","authors":["Atsuyuki Miyai","Zaiying Zhao","Kazuki Egashira","Atsuki Sato","Tatsumi Sunada","Shota Onohara","Hiromasa Yamanishi","Mashiro Toyooka","Kunato Nishina","Ryoma Maeda","Kiyoharu Aizawa","Toshihiko Yamasaki"],"authors_zh":"Atsuyuki Miyai、Zaiying Zhao、Kazuki Egashira、Atsuki Sato、Tatsumi Sunada、Shota Onohara、Hiromasa Yamanishi、Mashiro Toyooka、Kunato Nishina、Ryoma Maeda、Kiyoharu Aizawa、Toshihiko Yamasaki","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","data_release","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","search_substrate","reward_verifier_layer","release_audit"],"domains":["agent_trajectories","environment_interaction","web_browsing","long_horizon_tasks","memory_and_calculation"],"tags":["environment-agent-trajectory-data","web-agent-benchmark","browser-agent","webarena","long-horizon-tasks","memory-intensive-tasks","programmatic-evaluation","llm-judge","environment-reset","replay-risk"],"status":"partial","priority":"可读","paper_type_zh":"长程网页智能体任务与终端评测基准","best_for_zh":"研究网页智能体评测、长程记忆与计算任务、终端 verifier、环境重放和轨迹发布边界的读者","confidence":"high","one_line":["WebChoreArena releases 532 author-curated WebArena tasks with answer-, URL-, and DOM-based terminal evaluators, but not paper-run trajectories or a fully pinned environment snapshot.","WebChoreArena 发布 532 个作者策划的 WebArena 任务及答案、URL、DOM 终端评测器，但未发布论文实验的完整轨迹、结果包或完全固定的环境镜像。"],"why":"It turns memory-heavy browser chores into reusable task/evaluator records whose feedback can be inspected at the terminal boundary, while making clear that an executable benchmark is not automatically a trajectory dataset or a training release.","primary_link":"https://arxiv.org/abs/2506.01952","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/WebChoreArena/WebChoreArena"},{"key":"data","label":["Data","数据"],"url":"https://github.com/WebChoreArena/WebChoreArena/blob/542abc538fd9558362119714989166904d82f5f4/BrowserGym/config_files/test_webchore.raw.json"},{"key":"project","label":["Project","项目主页"],"url":"https://webchorearena.github.io/"}],"link_count":12,"sections":9},{"id":"webdancer-2025","title":"WebDancer: Towards Autonomous Information Seeking Agency","year":2025,"venue":"NeurIPS 2025","authors":["Jialong Wu","Baixuan Li","Runnan Fang","Wenbiao Yin","Liwen Zhang","Zhenglin Wang","Zhengwei Tao","Ding-Chu Zhang","Zekun Xi","Robert Tang","Yong Jiang","Pengjun Xie","Fei Huang","Jingren Zhou"],"authors_zh":"Jialong Wu, Baixuan Li, Runnan Fang, Wenbiao Yin, Liwen Zhang, Zhenglin Wang, Zhengwei Tao, Ding-Chu Zhang, Zekun Xi, Robert Tang, Yong Jiang, Pengjun Xie, Fei Huang, Jingren Zhou","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","model_report","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","environment_interaction","web_search","information_seeking"],"tags":["environment-agent-trajectory-data","web-agent","information-seeking","react-trajectories","synthetic-qa","trajectory-sft","agent-rl","llm-judge","live-web","partial-data-release"],"status":"partial","priority":"可读","paper_type_zh":"Web Agent 轨迹构造与 Agent RL 配方","best_for_zh":"研究 web-agent 数据合成、轨迹 SFT、judge-based agent RL 与 live-web 复现审计的读者","confidence":"high","one_line":["WebDancer turns 100,000 synthetic web QA pairs into filtered ReAct SFT trajectories and LLM-judged on-policy agent rollouts, but publicly releases only 200 QA and 200 trajectory examples rather than the complete corpus.","WebDancer 将 10 万条合成 web QA 转化为过滤后的 ReAct SFT 轨迹和由 LLM judge 评分的 on-policy rollout，但仅公开 200 条 QA 与 200 条冷启动轨迹样例。"],"why":"It exposes how task synthesis, trajectory teachers, observation masking, rejection filters, dynamic rollout selection, and judge-based terminal rewards jointly define web-agent post-training data, while showing why live-web drift and sample-only releases block record-level replay and audit.","primary_link":"https://papers.neurips.cc/paper_files/paper/2025/file/af043aee9cb137785d93195c6cf4cd96-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebDancer"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebDancer/datasets"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/Alibaba-NLP/WebDancer-32B"}],"link_count":12,"sections":9},{"id":"webevolver-2025","title":"WebEvolver: Enhancing Web Agent Self-Improvement with Co-evolving World Model","year":2025,"venue":"EMNLP 2025","authors":["Tianqing Fang","Hongming Zhang","Zhisong Zhang","Kaixin Ma","Wenhao Yu","Haitao Mi","Dong Yu"],"authors_zh":"Tianqing Fang、Hongming Zhang、Zhisong Zhang、Kaixin Ma、Wenhao Yu、Haitao Mi、Dong Yu","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe","data_release","agent_environment"],"verification_contract":["judgment_required","environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["sft","agent_training","test_time_compute"],"construction_layer":["trace_writing","self_play_anchor","reward_verifier_layer","optimizer_scaffold"],"domains":["web_navigation","gui_agents","information_seeking"],"tags":["webevolver","web-agent","self-improvement","world-model","synthetic-trajectories","live-web","cognitive-kernel","rejection-sampling","wmla","sft"],"status":"partial","priority":"可读","paper_type_zh":"推理数据、搜索或测试时扩展研究","best_for_zh":"需要核查推理轨迹、反馈契约、发布边界和复用风险的读者","confidence":"medium","one_line":["WebEvolver co-trains Llama-3.3-70B policy and world models, releases policy and world-model SFT message data, and uses synthetic rollouts plus world-model look-ahead; raw trajectories and replay metadata are not released.","WebEvolver 联合训练 Llama-3.3-70B 的策略模型与世界模型，发布了两者的监督微调消息数据，并使用合成采样与世界模型前瞻；原始轨迹与回放元数据未发布。"],"why":"It makes a world model both a training-data generator and an inference-time search component, creating a useful rollout-data case whose released SFT boundary must be kept separate from the paper's unreleased trajectory pipeline.","primary_link":"https://aclanthology.org/2025.emnlp-main.454/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Tencent/SelfEvolvingAgent/tree/main/WebEvolver"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/CognitiveKernel/WebEvolver"}],"link_count":5,"sections":9},{"id":"webexplorer-2025","title":"WebExplorer: Explore and Evolve for Training Long-Horizon Web Agents","year":2025,"venue":"arXiv preprint; submitted to ICLR 2026","authors":["Junteng Liu","Yunji Li","Chi Zhang","Jingyang Li","Aili Chen","Ke Ji","Weiyu Cheng","Zijia Wu","Chengyu Du","Qidi Xu","Jiayuan Song","Zhengmao Zhu","Wenhu Chen","Pengyu Zhao","Junxian He"],"authors_zh":"Junteng Liu, Yunji Li, Chi Zhang, Jingyang Li, Aili Chen, Ke Ji, Weiyu Cheng, Zijia Wu, Chengyu Du, Qidi Xu, Jiayuan Song, Zhengmao Zhu, Wenhu Chen, Pengyu Zhao, Junxian He","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","model_report","data_release"],"verification_contract":["mixed"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["agent_trajectories","environment_interaction","web_navigation","information_seeking"],"tags":["environment-agent-trajectory-data","web-agent","information-seeking","synthetic-qa","long-horizon","react-trajectory","sft","reinforcement-learning","llm-as-judge","partial-release"],"status":"partial","priority":"可读","paper_type_zh":"网页智能体数据构建、训练配方与部分数据发布","best_for_zh":"研究长程网页智能体数据合成、ReAct 轨迹 SFT、LLM judge 奖励与发布完整性审计的读者","confidence":"medium","one_line":["WebExplorer synthesizes about 40K evolved web-search QA pairs and trains Qwen3-8B with correct ReAct trajectories plus judge-rewarded online episodes, but releases only 100 QA rows rather than the paper-scale trajectories or training pipeline.","WebExplorer 先合成约 40K 条演化式网页搜索 QA，再用约 13K 条仅保留正确答案的 ReAct 轨迹做 SFT、用约 12K 条 QA 做 format 与 DeepSeek-V3 混合奖励的在线 RL；当前公开物只有 100 条 QA，没有论文规模轨迹或训练代码。"],"why":"It shows how hard-query construction, full-trajectory imitation, live search/browse observations, and answer-level model judgment interact in long-horizon agent training, while exposing why release shape, judge calibration, replayability, and benchmark-overlap evidence must be audited separately.","primary_link":"https://arxiv.org/abs/2509.06501","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/WebExplorer"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hkust-nlp/WebExplorer-QA"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/hkust-nlp/WebExplorer-8B"}],"link_count":6,"sections":9},{"id":"webgen-bench-2025","title":"WebGen-Bench: Evaluating LLMs on Generating Interactive and Functional Websites from Scratch","year":2025,"venue":"arXiv preprint","authors":["Zimu Lu","Yunqiao Yang","Houxing Ren","Haotian Hou","Han Xiao","Ke Wang","Weikang Shi","Aojun Zhou","Mingjie Zhan","Hongsheng Li"],"authors_zh":"Zimu Lu 等（Multimedia Laboratory, The Chinese University of Hong Kong）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["programmatic","environmental"],"supervision_granularity":["full_episode","answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["web-swe-agents","software-engineering-agent-benchmark"],"tags":["benchmark","software_engineering_agent_benchmark","web-swe-agents"],"status":"verified","priority":"可读","paper_type_zh":"arXiv 预印本的基准与评测面","best_for_zh":"关注基准、评分契约、污染风险、评测框架和后训练反馈可复用性的研究者。","confidence":"high","one_line":["WebGen-Bench exposes website generation with browser-executed functional tests as an auditable evaluation surface.","WebGen-Bench 将网站生成与浏览器执行的功能测试做成可审计的评测面。"],"why":"Evaluates building interactive web apps from scratch.","primary_link":"https://arxiv.org/abs/2505.03733","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mnluzimu/WebGen-Bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/luzimu/WebGen-Bench"}],"link_count":5,"sections":9},{"id":"webrl-2025","title":"WebRL: Training LLM Web Agents via Self-Evolving Online Curriculum Reinforcement Learning","year":2025,"venue":"ICLR 2025","authors":["Zehan Qi","Xiao Liu","Iat Long Iong","Hanyu Lai","Xueqiao Sun","Wenyi Zhao","Yu Yang","Xinyue Yang","Jiadai Sun","Shuntian Yao","Tianjie Zhang","Wei Xu","Jie Tang","Yuxiao Dong"],"authors_zh":"Zehan Qi、Xiao Liu、Iat Long Iong、Hanyu Lai、Xueqiao Sun、Wenyi Zhao、Yu Yang、Xinyue Yang、Jiadai Sun、Shuntian Yao、Tianjie Zhang、Wei Xu、Jie Tang、Yuxiao Dong","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward","trajectory_value"],"training_use":["sft","reward_modeling","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["web_agents","browser_interaction","reinforcement_learning"],"tags":["webrl","web-agent","webarena-lite","online-rl","outcome-reward-model","experience-replay","curriculum","trajectory-data"],"status":"partial","priority":"必读","paper_type_zh":"网页智能体在线课程强化学习、结果奖励模型与轨迹构造研究","best_for_zh":"关注浏览器智能体、在线强化学习、trajectory data、learned verifier、经验回放和发布审计的研究者与工程人员","confidence":"high","one_line":["WebRL trains browser agents over 8 curriculum phases using failure-derived GPT-4o tasks, binary ORM-labeled rollouts, KL-constrained updates, and confidence-filtered successful replay, while releasing only a partial data corpus.","WebRL 从 1,186 条 WebArena-Lite 种子示范出发，以失败驱动课程、学习型 ORM 和成功轨迹回放训练浏览器智能体；方法证据充分，但完整在线语料与复用许可仍未发布。"],"why":"It exposes how interactive agent trajectories become post-training records and shows that task generation, terminal reward modeling, and replay selection are inseparable parts of the feedback contract; it also illustrates why model/code release does not imply an auditable trajectory release.","primary_link":"https://openreview.net/forum?id=oVKEAFjEqv","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/WebRL/tree/fa8439ef1ea6ea2b327ee4162c8beaa309c53626"},{"key":"data","label":["Data","数据"],"url":"https://github.com/THUDM/WebRL/blob/fa8439ef1ea6ea2b327ee4162c8beaa309c53626/scripts/webarena_lite_sft.pt"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/zai-org/webrl-glm-4-9b/tree/dcc5fbea4b03cbc8448d781c5b2bc4dedbd7d5fb"}],"link_count":9,"sections":9},{"id":"websailor-v2-2025","title":"WebSailor-V2: Bridging the Chasm to Proprietary Agents via Synthetic Data and Scalable Reinforcement Learning","year":2025,"venue":"ICLR 2026 Poster","authors":["Kuan Li","Zhongwang Zhang","Huifeng Yin","Rui Ye","Yida Zhao","Liwen Zhang","Litu Ou","Ding-Chu Zhang","Xixi Wu","Xinmiao Yu","Jialong Wu","Xinyu Wang","Zile Qiao","Zhen Zhang","Yong Jiang","Pengjun Xie","Fei Huang","Zhi-Qin John Xu","Shuai Wang","Minhao Cheng","Jingren Zhou"],"authors_zh":"Kuan Li, Zhongwang Zhang, Huifeng Yin, Rui Ye, Yida Zhao, Liwen Zhang, Litu Ou, Ding-Chu Zhang, Xixi Wu, Xinmiao Yu, Jialong Wu, Xinyu Wang, Zile Qiao, Zhen Zhang, Yong Jiang, Pengjun Xie, Fei Huang, Zhi-Qin John Xu, Shuai Wang, Minhao Cheng, Jingren Zhou","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","model_report","agent_environment"],"verification_contract":["unknown"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","frontier_pipeline"],"domains":["agent_trajectories","environment_interaction","tool_use","information_seeking","web_search","question_answering"],"tags":["environment-agent-trajectory-data","web-agent","information-seeking","synthetic-data","dense-knowledge-graph","react-trajectories","rejection-sampling","on-policy-rl","simulated-environment","live-web-environment","undisclosed-verifier","partial-release","version-boundary"],"status":"partial","priority":"必读","paper_type_zh":"信息检索 agent 的合成数据构造与双环境 SFT/RL 方法","best_for_zh":"研究 web-agent 轨迹、合成 QA、双环境 agentic RL、release-scale 审计与失败轨迹保留的读者","confidence":"medium","one_line":["WebSailor-V2 reports 30K-plus synthetic instruction pairs, rejection-sampled ReAct SFT, and on-policy RL across offline-Wikipedia and managed live-web environments, but releases no V2 data, code, verifier, or complete trajectories.","WebSailor-V2 报告了 3 万余条合成 instruction pair、经 rejection sampling 获得的 ReAct SFT 轨迹，以及跨离线 Wikipedia 与受管真实网络环境的 on-policy RL，但没有公开 V2 数据、代码、verifier 或完整成败轨迹。"],"why":"It is a strong construction-recipe case for how topology-aware web tasks, successful tool-use traces, and dual-environment RL can be coupled, while sharply demonstrating why paper scale, public release scale, reward reproducibility, V1/V2 provenance, and failure retention must be audited independently.","primary_link":"https://openreview.net/pdf?id=HuP16O5SJf","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebSailor-V2"}],"link_count":13,"sections":9},{"id":"websailor-2025","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","year":2025,"venue":"arXiv preprint","authors":["Kuan Li","Zhongwang Zhang","Huifeng Yin","Liwen Zhang","Litu Ou","Jialong Wu","Wenbiao Yin","Baixuan Li","Zhengwei Tao","Xinyu Wang","Weizhou Shen","Junkai Zhang","Dingchu Zhang","Xixi Wu","Yong Jiang","Ming Yan","Pengjun Xie","Fei Huang","Jingren Zhou"],"authors_zh":"Kuan Li, Zhongwang Zhang, Huifeng Yin, Liwen Zhang, Litu Ou, Jialong Wu, Wenbiao Yin, Baixuan Li, Zhengwei Tao, Xinyu Wang, Weizhou Shen, Junkai Zhang, Dingchu Zhang, Xixi Wu, Yong Jiang, Ming Yan, Pengjun Xie, Fei Huang, Jingren Zhou","tracks":["environment_agent_trajectory_data"],"source_role":["construction_recipe","model_report","data_release","verifier_reward","agent_environment"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["agent_trajectories","environment_interaction","tool_use","information_seeking","web_search","question_answering"],"tags":["environment-agent-trajectory-data","web-agent","information-seeking","synthetic-data","reconstructed-trajectories","rejection-sampling","agentic-rl","mixed-verifier","llm-as-judge","partial-release","version-drift"],"status":"partial","priority":"必读","paper_type_zh":"复杂信息检索 agent 的数据构造与 RFT/RL 训练方法","best_for_zh":"研究 web agent、轨迹重构、agentic RL、混合 verifier 与训练数据发布审计的读者","confidence":"medium","one_line":["WebSailor trains web-search agents with just over 2,000 successful reconstructed ReAct trajectories and DUPO's mixed format/answer reward, but officially releases only 20 QA examples—not the complete successful or failed trajectories.","WebSailor 用 2,000 余条成功的重构 ReAct 轨迹和 DUPO 的混合格式/答案奖励训练网页搜索 agent，但官方目前只发布了 20 条 QA 示例，并未发布完整成功或失败轨迹。"],"why":"It links task difficulty, expert trajectory reconstruction, observation-masked RFT, and live-environment RL in one concrete post-training pipeline, while clearly showing why paper-reported dataset scale, public release scale, judge reproducibility, and failed-rollout retention must be tracked separately.","primary_link":"https://arxiv.org/abs/2507.02592","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebSailor"},{"key":"data","label":["Data","数据"],"url":"https://github.com/Alibaba-NLP/DeepResearch/blob/main/WebAgent/WebSailor/dataset/sailorfog-QA.jsonl"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/Alibaba-NLP/webagent"}],"link_count":12,"sections":9},{"id":"webserv-2025","title":"WEBSERV: A Full-Stack and RL-Ready Web Environment for Training Web Agents at Scale","year":2025,"venue":"arXiv preprint (expanded v2); earlier version presented at MTI-LLM @ NeurIPS 2025","authors":["Yuxuan Lu","Ziyi Wang","Jing Huang","Hui Liu","Jiri Gesi","Yan Han","Shihan Fu","Tianqi Zheng","Xianfeng Tang","Chen Luo","Yisi Sang","Jin Lai","Dakuo Wang"],"authors_zh":"Yuxuan Lu, Ziyi Wang, Jing Huang, Hui Liu, Jiri Gesi, Yan Han, Shihan Fu, Tianqi Zheng, Xianfeng Tang, Chen Luo, Yisi Sang, Jin Lai, Dakuo Wang","tracks":["environment_agent_trajectory_data"],"source_role":["agent_environment","infrastructure","data_release","construction_recipe","verifier_reward","scaling_study"],"verification_contract":["mixed","environmental"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["sft","rlvr","agent_training","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","scaling_report","release_audit"],"domains":["web_agents","agent_trajectories","browser_interaction","environment_interaction","reinforcement_learning","ecommerce","content_management","software_development"],"tags":["environment-agent-trajectory-data","web-agent","browser-interaction","webarena-lite","incus","container-isolation","resettable-environment","sft-trajectories","on-policy-rl","grpo","environmental-verifier","release-gap"],"status":"partial","priority":"必读","paper_type_zh":"网页智能体环境、轨迹数据发布与强化学习训练基础设施","best_for_zh":"研究网页智能体 SFT/RL 数据、环境重置、浏览器动作可靠性、环境奖励以及论文规模与实际发布差异的读者","confidence":"medium","one_line":["WebServ releases 726 success-filtered Claude browser traces and the code for resettable Incus-backed GRPO training, but not failed traces, record-level outcomes, Qwen RL rollouts, checkpoints, or a pinned environment bundle.","WebServ 公开了 726 条经成功筛选的 Claude 浏览器轨迹及 Incus 隔离式 GRPO 训练代码，但未公开失败轨迹、逐条结果标签、Qwen 在线 RL rollout、模型检查点或可固定复现的环境镜像。"],"why":"It connects trajectory quality to the full browser-server stack: compact observations, network-aware actions, isolated resettable applications, executable WebArena rewards, and rollout throughput. It also demonstrates why reported rollout volume must be audited separately from public data retention.","primary_link":"https://arxiv.org/abs/2510.16252","links":[{"key":"code","label":["Code","代码"],"url":"https://anonymous.4open.science/r/webserv_anonymous-90EF/"}],"link_count":9,"sections":9},{"id":"when-data-is-the-algorithm-a-systematic-study-and-curation-of-preference-optimization-datasets-2025","title":"When Data is the Algorithm: A Systematic Study and Curation of Preference Optimization Datasets","year":2025,"venue":"arXiv","authors":["Aladin Djuhera","Farhan Ahmed","Swanand Ravindra Kadhe","Syed Zawad","Heiko Ludwig","Holger Boche"],"authors_zh":"Aladin Djuhera、Farhan Ahmed、Swanand Ravindra Kadhe、Syed Zawad、Heiko Ludwig、Holger Boche","tracks":["preference_reward_feedback_data"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["When Data is the Algorithm: A Systematic Study and Curation of Preference Optimization Datasets addresses curation and systematic study of preference-optimization data.","该工作分析开放 DPO 语料，并利用逐样本任务、输入质量和奖励标注构建 UltraMix。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2511.10985","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/aladinDJ/ultramix-DPO-annotated"}],"link_count":2,"sections":9},{"id":"solve-verify-compute-optimal-2025","title":"When To Solve, When To Verify: Compute-Optimal Problem Solving and Generative Verification for LLM Reasoning","year":2025,"venue":"COLM 2025","authors":["Nishad Singhi","Hritik Bansal","Arian Hosseini","Aditya Grover","Kai-Wei Chang","Marcus Rohrbach","Anna Rohrbach"],"authors_zh":"Nishad Singhi、Hritik Bansal、Arian Hosseini、Aditya Grover、Kai-Wei Chang、Marcus Rohrbach、Anna Rohrbach（机构：达姆施塔特工业大学、hessian.AI、加州大学洛杉矶分校、Google DeepMind、Mila）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["scaling_report","optimizer_scaffold"],"domains":["mathematical-reasoning","scientific-reasoning"],"tags":["test-time-compute","generative-verification","self-consistency","scaling-laws","compute-allocation"],"status":"verified","priority":"必读","paper_type_zh":"面向求解与生成式验证的计算最优测试时扩展分析","best_for_zh":"为数学或科学推理设计验证器感知推理预算的读者。","confidence":"high","one_line":["The paper shows when solution sampling beats generative verification and fits scaling laws for allocating compute between them.","论文说明何时应扩大解答采样、何时应扩大生成式验证，并拟合二者的计算分配标度律。"],"why":"It measures verification cost rather than treating a stronger verifier as free, yielding an actionable solve-versus-verify allocation rule.","primary_link":"https://arxiv.org/abs/2504.01005","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/nishadsinghi/sc-genrm-scaling"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/sc-genrm-scaling"}],"link_count":4,"sections":9},{"id":"agentdebug-failure-trajectories-2025","title":"Where LLM Agents Fail and How They can Learn From Failures","year":2025,"venue":"arXiv preprint","authors":["Kunlun Zhu","Zijia Liu","Bingxuan Li","Muxin Tian","Yingxuan Yang","Jiaxun Zhang","Pengrui Han","Qipeng Xie","Fuyang Cui","Weijia Zhang","Xiaoteng Ma","Xiaodong Yu","Gowtham Ramesh","Jialian Wu","Zicheng Liu","Pan Lu","James Zou","Jiaxuan You"],"authors_zh":"Kunlun Zhu、Zijia Liu、Bingxuan Li、Muxin Tian、Yingxuan Yang、Jiaxun Zhang、Pengrui Han、Qipeng Xie、Fuyang Cui、Weijia Zhang、Xiaoteng Ma、Xiaodong Yu、Gowtham Ramesh、Jialian Wu、Zicheng Liu、Pan Lu、James Zou、Jiaxuan You","tracks":["rollout_search_test_time_trace_data","process_trace_supervision_data"],"source_role":["benchmark","audit_failure","agent_environment"],"verification_contract":["environmental","judgment_required","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate"],"domains":["reasoning"],"tags":["agent-failures","failure-trajectories","debugging","error-taxonomy"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["AgentErrorBench releases 200 annotated failed ALFWorld, WebShop, and GAIA episodes with a 17-type taxonomy, minimal root causes, and feedback, while annotation agreement and replay metadata remain material audit limits.","AgentErrorBench 发布 200 条带标注的 ALFWorld、WebShop 与 GAIA 失败轨迹，含 17 类错误分类、最小根因与纠正反馈；标注一致性与回放元数据仍是主要审计缺口。"],"why":"It turns failed agent rollouts into structured supervision for step/module detection, causal localization, feedback generation, and recovery, preserving the distinction between environment success and judgment-based diagnosis.","primary_link":"https://arxiv.org/abs/2509.25370","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ulab-uiuc/AgentDebug"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/drive/folders/1bQe6dQA85pktT63YnKIKJDTVaH3O3Vpu?usp=drive_link"}],"link_count":5,"sections":9},{"id":"mast-multi-agent-system-failure-taxonomy-2025","title":"Why Do Multi-Agent LLM Systems Fail?","year":2025,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Mert Cemri","Melissa Z. Pan","Shuyi Yang","Lakshya A. Agrawal","Bhavya Chopra","Rishabh Tiwari","Kurt Keutzer","Aditya Parameswaran","Dan Klein","Kannan Ramchandran","Matei Zaharia","Joseph E. Gonzalez","Ion Stoica"],"authors_zh":"Mert Cemri、Melissa Z. Pan、Shuyi Yang、Lakshya A. Agrawal、Bhavya Chopra、Rishabh Tiwari、Kurt Keutzer、Aditya Parameswaran、Dan Klein、Kannan Ramchandran、Matei Zaharia、Joseph E. Gonzalez、Ion Stoica","tracks":["process_trace_supervision_data"],"source_role":["data_release","benchmark","process_supervision","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["multi-agent-systems","failure-taxonomy","trace-diagnosis"],"tags":["process-supervision","agent-trajectories","reasoning"],"status":"verified","priority":"必读","paper_type_zh":"过程与轨迹监督数据论文","best_for_zh":"需要依据统一失效分类分析、训练或评测多智能体协作轨迹诊断器的研究者。","confidence":"high","one_line":["MAST-Data scales a human-derived 14-mode failure taxonomy to more than 1,600 multi-agent execution traces for structured diagnosis and judge evaluation.","MAST-Data 将人工归纳的 14 类多智能体失效分类扩展到逾 1,600 条执行轨迹，支持协作过程诊断与 judge 评测。"],"why":"It turns trace diagnosis into a reusable taxonomy-aligned supervision problem across multiple multi-agent frameworks.","primary_link":"https://arxiv.org/abs/2503.13657","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/multi-agent-systems-failure-taxonomy/MAST"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mcemri/MAST-Data"},{"key":"project","label":["Project","项目主页"],"url":"https://sites.google.com/berkeley.edu/mast/"}],"link_count":4,"sections":9},{"id":"wider-or-deeper-ab-mcts-2025","title":"Wider or Deeper? Scaling LLM Inference-Time Compute with Adaptive Branching Tree Search","year":2025,"venue":"NeurIPS 2025","authors":["Yuichi Inoue","Kou Misaki","Yuki Imajuku","So Kuroki","Taishi Nakamura","Takuya Akiba"],"authors_zh":"Yuichi Inoue、Kou Misaki、Yuki Imajuku、So Kuroki、Taishi Nakamura、Takuya Akiba","tracks":["rollout_search_test_time_trace_data","scaling_rlvr_test_time_compute"],"source_role":["construction_recipe","scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","scalar_reward","trajectory_value"],"training_use":["test_time_compute","evaluation"],"construction_layer":["search_substrate","reward_verifier_layer","scaling_report"],"domains":["coding","software_engineering","abstract_reasoning","machine_learning_engineering","reasoning"],"tags":["ab-mcts","adaptive-branching","monte-carlo-tree-search","response-revision-trees","external-feedback","test-time-compute","treequest","coding"],"status":"partial","priority":"可读","paper_type_zh":"自适应分支测试时树搜索配方与推理计算扩展研究","best_for_zh":"研究测试时计算、搜索轨迹数据、外部反馈分配、响应修订谱系、预算归因与原始树发布审计的读者","confidence":"high","one_line":["AB-MCTS allocates a fixed inference budget between fresh response branches and feedback-conditioned revisions, producing scored answer trees that released code can persist even though the paper's realized trees are not public.","AB-MCTS 在固定推理预算下，自适应选择生成新回答分支或沿既有回答进行反馈驱动修订，形成带外部评分的响应—修订树；公开代码可保存完整本地树和逐调用日志，但论文实验的原始树、全部回答、未选分支与 API 日志并未发布。"],"why":"It turns test-time compute into an auditable trace schema—prompts, candidate answers, revisions, scores, tree edges, and budgets—while showing why open algorithms alone are insufficient without raw trees, rejected branches, evaluator context, and stable API revisions.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/32dff2e27b0da1fb5f4209216a948544-Paper-Conference.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SakanaAI/treequest"},{"key":"project","label":["Project","项目主页"],"url":"https://sakana.ai/ab-mcts/"}],"link_count":7,"sections":9},{"id":"wildbench-real-user-evaluation-2024","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","year":2025,"venue":"arXiv preprint","authors":["Bill Yuchen Lin","Yuntian Deng","Khyathi Chandu","Faeze Brahman","Abhilasha Ravichander","Valentina Pyatkin","Nouha Dziri","Ronan Le Bras","Yejin Choi"],"authors_zh":"Bill Yuchen Lin、Yuntian Deng、Khyathi Chandu、Faeze Brahman、Abhilasha Ravichander、Valentina Pyatkin、Nouha Dziri、Ronan Le Bras、Yejin Choi","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["WildBench curates hard real-user tasks and applies task-specific checklists for automatic, interpretable model comparison.","WildBench 从真实用户对话筛选高难任务，并用任务专属核对表进行自动、可解释的模型比较。"],"why":"WildBench curates hard real-user tasks and applies task-specific checklists for automatic, interpretable model comparison.","primary_link":"https://arxiv.org/abs/2406.04770","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/WildBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/WildBench"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/allenai/WildBench"}],"link_count":4,"sections":9},{"id":"womd-reasoning-2025","title":"WOMD-Reasoning: A Large-Scale Dataset for Interaction Reasoning in Driving","year":2025,"venue":"ICML 2025","authors":["Yiheng Li","Cunxin Fan","Chongjian Ge","Seth Z. Zhao","Chenran Li","Chenfeng Xu","Huaxiu Yao","Masayoshi Tomizuka","Bolei Zhou","Chen Tang","Mingyu Ding","Wei Zhan"],"authors_zh":"Yiheng Li, Cunxin Fan, Chongjian Ge, Seth Z. Zhao, Chenran Li, Chenfeng Xu, Huaxiu Yao, Masayoshi Tomizuka, Bolei Zhou, Chen Tang, Mingyu Ding, Wei Zhan","tracks":["instruction_demonstration_rationale_data","environment_agent_trajectory_data"],"source_role":["data_release","benchmark","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["autonomous_driving","motion","traffic_interaction","intention_reasoning","multimodal"],"tags":["womd","waymo","driving","interaction-reasoning","intention-reasoning","synthetic-qa"],"status":"partial","priority":"可读","paper_type_zh":"驾驶交互推理数据集与构造配方","best_for_zh":"研究驾驶语言监督、交互推理和 hindsight 泄漏的读者","confidence":"medium","one_line":["WOMD-Reasoning converts 63,000 WOMD scenes into 2.94 million GPT-4 Turbo Q&As, including 409,000 interaction and intention records that may depend on future trajectories.","WOMD-Reasoning 将 6.3 万个 Waymo 运动场景转成 294 万条规则支撑与 LLM 撰写的问答，但生成式交互叙述和未来感知标签必须分别进行因果与质量审计。"],"why":"It is a large driving-language supervision object with explicit scene lineage and released construction prompts, but reuse requires separating factual from generated reasoning, marking future-conditioned targets, and auditing the unverified majority of records.","primary_link":"https://proceedings.mlr.press/v267/li25l.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yhli123/WOMD-Reasoning"},{"key":"data","label":["Data","数据"],"url":"https://waymo.com/open/download/"}],"link_count":5,"sections":9},{"id":"xverify-efficient-answer-verifier-2025","title":"xVerify: Efficient Answer Verifier for Reasoning Model Evaluations","year":2025,"venue":"arXiv 2025","authors":["Ding Chen","Qingchen Yu","Pengyuan Wang","Wentao Zhang","Bo Tang","Feiyu Xiong","Xinchi Li","Minchuan Yang","Zhiyu Li"],"authors_zh":"Ding Chen, Qingchen Yu, Pengyuan Wang, Wentao Zhang, Bo Tang, Feiyu Xiong, Xinchi Li, Minchuan Yang, Zhiyu Li","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["reasoning-evaluation","answer-verification","reward-modeling"],"tags":["preference-data","reward-modeling","reasoning"],"status":"verified","priority":"可读","paper_type_zh":"答案验证器与带正确性反馈的数据集论文","best_for_zh":"需要训练轻量答案验证器，或审计复杂推理输出是否等价于参考答案的研究者。","confidence":"high","one_line":["xVerify trains lightweight answer-equivalence verifiers on VAR, a roughly 56K-record collection of long reasoning answers with correctness labels.","约5.6万问答验证样本以参考答案给出对错标签，适用于训练轻量验证器与推理结果评测。"],"why":"The release makes answer-level feedback and its reference-based acceptance rule reusable for reasoning evaluation and reward modeling.","primary_link":"https://arxiv.org/abs/2504.10481","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/IAAR-Shanghai/xVerify"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/IAAR-Shanghai/VAR"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/IAAR-Shanghai/xverify-67e0f6f94c2dc334727da802"}],"link_count":6,"sections":9},{"id":"yescieval-2025","title":"YESciEval: Robust LLM-as-a-Judge for Scientific Question Answering","year":2025,"venue":"ACL 2025","authors":["Jennifer D’Souza","Hamed Babaei Giglou","Quentin Münch"],"authors_zh":"Jennifer D’Souza、Hamed Babaei Giglou、Quentin Münch","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","infrastructure"],"verification_contract":["judgment_required","mixed"],"supervision_granularity":["answer_level"],"training_use":["reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["llm-as-a-judge","rubric","evaluation-reliability"],"tags":["track07","judgment-rubric","2025-2026"],"status":"verified","priority":"可读","paper_type_zh":"科学问答 LLM-as-a-judge 评测、对抗审计与对齐论文","best_for_zh":"需要审计科学问答 judge 或构建 rubric 条件化评测集的研究者。","confidence":"medium","one_line":["A rubric-based evaluation framework for scientific question answering that audits judge robustness under adversarial data.","以 9 项 rubric、对抗变体和开源 judge 审计科学问答回答的可靠性。"],"why":"It provides an auditable judgment-required feedback surface for post-training reasoning data and evaluation.","primary_link":"https://aclanthology.org/2025.acl-long.675/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sciknoworg/YESciEval"},{"key":"data","label":["Data","数据"],"url":"https://data.uni-hannover.de/dataset/yescieval-corpus"}],"link_count":4,"sections":9},{"id":"z1-efficient-test-time-scaling-code-2025","title":"Z1: Efficient Test-time Scaling with Code","year":2025,"venue":"EMNLP 2025 Industry Track","authors":["Zhaojian Yu","Yinghao Wu","Yilun Zhao","Arman Cohan","Xiao-Ping Zhang"],"authors_zh":"Zhaojian Yu、Yinghao Wu、Yilun Zhao、Arman Cohan、Xiao-Ping Zhang","tracks":["rollout_search_test_time_trace_data"],"source_role":["data_release","construction_recipe","scaling_study"],"verification_contract":["unknown"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","optimizer_scaffold","scaling_report"],"domains":["code","reasoning"],"tags":["z1","code-reasoning","long-short-trajectories","teacher-distillation","shifted-thinking-window","test-time-compute","reasoning-efficiency","sft"],"status":"partial","priority":"可读","paper_type_zh":"代码推理轨迹发布、长短轨迹蒸馏与测试时预算研究","best_for_zh":"研究 rollout 轨迹、长短推理蒸馏、SFT 数据配方和 test-time compute 归因的读者","confidence":"high","one_line":["Z1 releases 107,173 Code Evol-Instruct-derived prompts with QwQ-32B-Preview response trajectories and pairs their SFT use with a token-capped Shifted Thinking Window for efficient inference.","Z1 发布 107,173 条 Code Evol-Instruct 派生问题与 QwQ-32B-Preview 完整响应轨迹，并用 Shifted Thinking Window 约束推理预算；但数据没有正确性 verifier、长短标签、许可证或去污染记录。"],"why":"It makes the interaction between training-trace length and inference-time reasoning budget unusually concrete, while its absent verifier, short/long labels, generation/filtering code, license, and decontamination record define clear audit boundaries.","primary_link":"https://aclanthology.org/2025.emnlp-industry.182/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/efficientscaling/Z1"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/efficientscaling/Z1-Code-Reasoning-107K"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/efficientscaling/Z1-7B"}],"link_count":6,"sections":9},{"id":"zebra-cot-2025","title":"Zebra-CoT: A Dataset for Interleaved Vision-Language Reasoning","year":2025,"venue":"arXiv preprint (2025)","authors":["Ang Li","Charles Wang","Deqing Fu","Kaiyu Yue","Zikui Cai","Wang Bill Zhu","Ollie Liu","Peng Guo","Willie Neiswanger","Furong Huang","Tom Goldstein","Micah Goldblum"],"authors_zh":"Ang Li、Charles Wang、Deqing Fu、Kaiyu Yue、Zikui Cai、Wang Bill Zhu、Ollie Liu、Peng Guo、Willie Neiswanger、Furong Huang、Tom Goldstein、Micah Goldblum","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["geometry, physics, algorithms, 2D and 3D reasoning, embodied planning, logic, games, and visual search"],"tags":["instruction-demonstration-rationale","arxiv-2507.16746","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放后训练数据集与构造研究","best_for_zh":"视觉思维链监督微调","confidence":"high","one_line":["Zebra-CoT serializes both textual thoughts and intermediate images so a model can learn to generate visual aids during multi-step reasoning.","Zebra-CoT 把文字思考与中间图像交错保存为 18.2 万条视觉推理轨迹，覆盖 18 个领域。"],"why":"Text-only CoT cannot demonstrate sketching, visual search, or intermediate spatial transformations, and off-the-shelf VLMs are too weak to bootstrap such traces reliably.","primary_link":"https://arxiv.org/abs/2507.16746","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/multimodal-reasoning-lab/Zebra-CoT"}],"link_count":2,"sections":9},{"id":"zip-rc-introspection-2025","title":"Zero-Overhead Introspection for Adaptive Test-Time Compute","year":2025,"venue":"ICLR 2026","authors":["Rohin Manvi","Joey Hong","Tim Seyde","Maxime Labonne","Mathias Lechner","Sergey Levine"],"authors_zh":"Rohin Manvi、Joey Hong、Tim Seyde、Maxime Labonne、Mathias Lechner、Sergey Levine（机构：加州大学伯克利分校、MIT CSAIL、Liquid AI）","tracks":["scaling_rlvr_test_time_compute"],"source_role":["scaling_study","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["token_level","full_episode"],"training_use":["test_time_compute","evaluation"],"construction_layer":["optimizer_scaffold","scaling_report"],"domains":["mathematical-reasoning"],"tags":["test-time-compute","introspection","adaptive-search","verifier","prefix-pruning"],"status":"verified","priority":"必读","paper_type_zh":"零开销内省与自适应测试时搜索研究（ICLR 2026）","best_for_zh":"构建同时控制答案质量、计算量和延迟的自适应并行解码系统的读者。","confidence":"high","one_line":["ZIP-RC reuses reserved logits to predict reward and remaining length without an extra inference pass, then allocates prefix-tree search by utility.","ZIP-RC 在同一次前向中预测奖励与剩余长度，并据此分配测试时搜索。"],"why":"It makes the reward–cost estimate itself an in-model inference object instead of paying for a separate verifier pass at every search decision.","primary_link":"https://arxiv.org/abs/2512.01457","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rohinmanvi/ZIP-RC"},{"key":"project","label":["Project","项目主页"],"url":"https://rohinmanvi.github.io/ZIP-RC/"}],"link_count":5,"sections":9},{"id":"llm-confidence-calibration-survey-2024","title":"A Survey of Confidence Estimation and Calibration in Large Language Models","year":2024,"venue":"NAACL 2024 Long Papers","authors":["Jiahui Geng","Fengyu Cai","Yuxia Wang","Heinz Koeppl","Preslav Nakov","Iryna Gurevych"],"authors_zh":"Jiahui Geng 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["confidence-estimation","calibration","factuality","evaluation"],"tags":["survey","confidence-estimation","calibration","factuality","naacl-2024"],"status":"verified","priority":"可读","paper_type_zh":"综述","best_for_zh":"需要设计选择性回答、置信度展示或可靠性评测的读者。","confidence":"high","one_line":["A NAACL 2024 survey of estimating and calibrating how reliable an LLM's answer is likely to be.","系统梳理大语言模型的置信度估计与校准：如何让自信程度更接近实际正确率。"],"why":"It makes confidence a measurable reliability question rather than treating a model's fluent style as evidence of correctness.","primary_link":"https://aclanthology.org/2024.naacl-long.366/","links":[],"link_count":2,"sections":9},{"id":"llm-evaluation-survey-2024","title":"A Survey on Evaluation of Large Language Models","year":2024,"venue":"ACM Transactions on Intelligent Systems and Technology","authors":["Yupeng Chang","Xu Wang","Jindong Wang","Yuan Wu","Linyi Yang","Kaijie Zhu","Hao Chen","Xiaoyuan Yi","Cunxiang Wang","Yidong Wang","Wei Ye","Yue Zhang","Yi Chang","Philip S. Yu","Qiang Yang","Xing Xie"],"authors_zh":"Yupeng Chang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["evaluation","reasoning","benchmarks","reasoning-data","reliability"],"tags":["foundations-and-primers","llm-evaluation","reasoning-benchmarks","survey","high-citation"],"status":"verified","priority":"必读","paper_type_zh":"大语言模型评测综述","best_for_zh":"设计评测套件或解读推理基准分数的读者。","confidence":"high","one_line":["A high-impact survey that turns “evaluate an LLM” into concrete choices about target, dataset, protocol, and metric.","把大模型评测拆成评什么、在哪里评、怎样评的高影响力综述。"],"why":"It prevents reasoning-data projects from treating one benchmark score as a complete capability claim.","primary_link":"https://dl.acm.org/doi/10.1145/3641289","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MLGroupJLU/LLM-eval-survey"}],"link_count":4,"sections":9},{"id":"llm-compression-survey-2024","title":"A Survey on Model Compression for Large Language Models","year":2024,"venue":"Transactions of the Association for Computational Linguistics","authors":["Xunyu Zhu","Jian Li","Yong Liu","Can Ma","Weiping Wang"],"authors_zh":"Xunyu Zhu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["model-compression","efficiency","llm-deployment"],"tags":["survey","compression","quantization","pruning","distillation","tacl-2024"],"status":"verified","priority":"可读","paper_type_zh":"大语言模型压缩综述","best_for_zh":"为受限部署选择或评估模型压缩方法的读者。","confidence":"high","one_line":["A TACL survey of quantization, pruning, and distillation for practical LLM deployment.","综述量化、剪枝和知识蒸馏等压缩方法，以及相应的评测指标。"],"why":"It frames compression as a trade-off between resource use and task quality rather than a single speed-up claim.","primary_link":"https://aclanthology.org/2024.tacl-1.85/","links":[],"link_count":2,"sections":9},{"id":"agentbank-2024","title":"AGENT BANK: Towards Generalized LLM Agents via Fine-Tuning on 50000+ Interaction Trajectories","year":2024,"venue":"Findings of EMNLP 2024","authors":["Yifan Song","Weimin Xiong","Xiutian Zhao","Dawei Zhu","Wenhao Wu","Ke Wang","Cheng Li","Wei Peng","Sujian Li"],"authors_zh":"Yifan Song, Weimin Xiong, Xiutian Zhao, Dawei Zhu, Wenhao Wu, Ke Wang, Cheng Li, Wei Peng, Sujian Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["environmental","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["agent-reasoning","mathematics","programming","web","embodied-ai","tool-use"],"tags":["instruction-demonstration-rationale","agent-trajectories","answer-forcing","samoyed","arxiv-2403.12881"],"status":"verified","priority":"必读","paper_type_zh":"大规模多任务代理轨迹数据集与监督微调研究","best_for_zh":"适合研究代理轨迹构造、答案强制、难度偏差和跨任务泛化的数据策展者。","confidence":"high","one_line":["AGENT BANK releases 51,287 paper-reported rationale-action-observation trajectories across 16 tasks for masked-loss trajectory tuning.","AGENT BANK 发布论文所报 51,287 条跨十六任务的理由、动作与观察轨迹，用于掩码式轨迹微调。"],"why":"It scales agent demonstrations while separating action recovery from rationale writing and explicitly studies difficulty bias and cross-task transfer.","primary_link":"https://aclanthology.org/2024.findings-emnlp.116/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Solaris99/AgentBank"}],"link_count":5,"sections":9},{"id":"agent-flan-2024","title":"Agent-FLAN: Designing Data and Methods of Effective Agent Tuning for Large Language Models","year":2024,"venue":"Findings of ACL 2024","authors":["Zehui Chen","Kuikun Liu","Qiuchen Wang","Wenwei Zhang","Jiangning Liu","Dahua Lin","Kai Chen","Feng Zhao"],"authors_zh":"Zehui Chen, Kuikun Liu, Qiuchen Wang, Wenwei Zhang, Jiangning Liu, Dahua Lin, Kai Chen, Feng Zhao","tracks":["instruction_demonstration_rationale_data","process_trace_supervision_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["trace_writing","release_audit"],"domains":["agent-reasoning","tool-use","web","embodied-ai","database","operating-systems"],"tags":["instruction-demonstration-rationale","agent-trajectories","tool-use","negative-supervision","arxiv-2403.12881"],"status":"verified","priority":"必读","paper_type_zh":"代理训练数据重设计与工具使用监督研究","best_for_zh":"适合研究代理数据表达、能力混合权重、工具调用幻觉和负例监督的读者。","confidence":"high","one_line":["Agent-FLAN releases 24,703 redesigned agent conversations that separate reasoning, retrieval, arguments, format following, and tool-use negatives for SFT.","Agent-FLAN 把代理轨迹改写为按能力配比的自然对话，并加入不应调用工具的负例，共公开 24,703 条 SFT 记录。"],"why":"It treats representation, capability balance, and non-action examples as data decisions and measures their effects separately from simply adding more trajectories.","primary_link":"https://aclanthology.org/2024.findings-acl.557/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InternLM/Agent-FLAN"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/internlm/Agent-FLAN"},{"key":"project","label":["Project","项目主页"],"url":"https://internlm.github.io/Agent-FLAN/"}],"link_count":7,"sections":9},{"id":"agentbank-generalized-llm-agents-2024","title":"AGENTBANK: Towards Generalized LLM Agents via Fine-Tuning on 50000+ Interaction Trajectories","year":2024,"venue":"Findings of EMNLP 2024","authors":["Yifan Song","Weimin Xiong","Xiutian Zhao","Dawei Zhu","Wenhao Wu","Ke Wang","Cheng Li","Wei Peng","Sujian Li"],"authors_zh":"Yifan Song, Weimin Xiong, Xiutian Zhao, Dawei Zhu, Wenhao Wu, Ke Wang, Cheng Li, Wei Peng, Sujian Li","tracks":["process_trace_supervision_data"],"source_role":["data_release","construction_recipe","process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level","process_reward"],"training_use":["process_supervision","reward_modeling","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer"],"domains":["agent-reasoning","interactive-environments","process-supervision"],"tags":["process-supervision","trace-data","reward-modeling"],"status":"verified","priority":"可读","paper_type_zh":"跨任务智能体交互轨迹与逐步思维标注数据集论文","best_for_zh":"需要大规模、跨环境 ReAct/CoT 轨迹来研究通用智能体微调和能力迁移的研究者。","confidence":"high","one_line":["AGENTBANK releases 50K+ cross-task interaction trajectories with chain-of-thought rationale for every action.","AGENTBANK 汇集 16 项任务、五类智能体技能的五万余条交互轨迹，并为每个动作提供链式推理标注。"],"why":"It is a widely reusable large-scale baseline for generalist agent trajectory tuning across multiple skill dimensions.","primary_link":"https://aclanthology.org/2024.findings-emnlp.116.pdf","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Solaris99/AgentBank"}],"link_count":5,"sections":9},{"id":"agentboard-2024","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","year":2024,"venue":"NeurIPS 2024 Oral / arXiv","authors":["Chang Ma","Junlei Zhang","Zhihao Zhu","Cheng Yang","Yujiu Yang","Yaohui Jin","Zhenzhong Lan","Lingpeng Kong","Junxian He"],"authors_zh":"Chang Ma、Junlei Zhang、Zhihao Zhu、Cheng Yang、Yujiu Yang、Yaohui Jin 等（The University of Hong Kong 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","audit_failure"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["tool-use","agent-environments","interactive-evaluation"],"tags":["tool-use","agent-environment","trajectory-data"],"status":"verified","priority":"可读","paper_type_zh":"Agent 环境 / 工具使用数据与评测","best_for_zh":"研究 tool-use、agent trajectory、环境反馈契约和可复现评测的读者","confidence":"medium","one_line":["AgentBoard contributes a tool-use or agent-environment surface with an explicit evaluation or verification contract.","AgentBoard 提供带明确评测契约的智能体环境评测面，除任务成功率外还用进度指标刻画多轮交互中的逐步推进。"],"why":"It makes the action schema, environment state, and success predicate visible enough for reasoning-data curation and audit.","primary_link":"https://arxiv.org/abs/2401.13178","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/AgentBoard"}],"link_count":3,"sections":9},{"id":"agenttuning-2024","title":"AgentTuning: Enabling Generalized Agent Abilities for LLMs","year":2024,"venue":"Findings of ACL 2024","authors":["Aohan Zeng","Mingdao Liu","Rui Lu","Bowen Wang","Xiao Liu","Yuxiao Dong","Jie Tang"],"authors_zh":"Aohan Zeng, Mingdao Liu, Rui Lu, Bowen Wang, Xiao Liu, Yuxiao Dong, Jie Tang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["environmental","mixed"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["agent-reasoning","tool-use","web","embodied-ai","database","operating-systems"],"tags":["instruction-demonstration-rationale","agent-trajectories","react","tool-use","arxiv-2310.12823"],"status":"verified","priority":"必读","paper_type_zh":"多环境代理示范数据集与监督微调研究","best_for_zh":"适合构建文本代理 SFT、审计思考—动作—观察记录和研究跨环境迁移的读者。","confidence":"high","one_line":["AgentInstruct releases 1,866 reward-filtered, multi-turn ReAct demonstrations across six agent tasks for hybrid instruction tuning.","AgentInstruct 将六类代理环境整理成 1,866 条按奖励筛选的多轮 ReAct 示范，用于与通用对话混合微调。"],"why":"It turns interactive thought-action-observation behavior into a small, inspectable SFT object and tests whether that object transfers beyond its source environments.","primary_link":"https://aclanthology.org/2024.findings-acl.181/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/AgentTuning"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zai-org/AgentInstruct"},{"key":"project","label":["Project","项目主页"],"url":"https://thudm.github.io/AgentTuning/"}],"link_count":7,"sections":9},{"id":"aider-polyglot-2025","title":"Aider's Polyglot Coding Benchmark","year":2024,"venue":"Aider official benchmark release","authors":["Aider"],"authors_zh":"Aider","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["full_episode","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["coding","multilingual-programming"],"tags":["benchmark","coding","evaluation-surface","polyglot"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"关注多语言代码编辑、仓库级补丁生成和测试驱动代码评测的读者。","confidence":"high","one_line":["Aider Polyglot evaluates coding-edit ability across C++, Go, Java, JavaScript, Python, and Rust using selected Exercism problems and repository-level test feedback.","Aider Polyglot 使用筛选后的 Exercism 题目，评测模型在 C++、Go、Java、JavaScript、Python 和 Rust 中完成代码编辑并通过测试的能力。"],"why":"Aider Polyglot evaluates coding-edit ability across C++, Go, Java, JavaScript, Python, and Rust using selected Exercism problems and repository-level test feedback.","primary_link":"https://aider.chat/2024/12/21/polyglot.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Aider-AI/polyglot-benchmark"}],"link_count":2,"sections":9},{"id":"aime-benchmark-2024","title":"AIME Benchmark for Mathematical Reasoning","year":2024,"venue":"Mathematical Association of America exam source / benchmark subset","authors":["Mathematical Association of America"],"authors_zh":"Mathematical Association of America 等（Mathematical Association of America）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["competition-math","olympiad-style-evaluation","contamination-audit"],"tags":["benchmark","competition-math","olympiad-style-evaluation","contamination-audit"],"status":"partial","priority":"暂缓","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"medium","one_line":["AIME 是高难度竞赛数学评测源，题量不大但区分度高，常用来观察 frontier reasoning 的上限。","AIME 是高难度竞赛数学评测源，题量不大但区分度高，常用来观察 frontier reasoning 的上限。"],"why":"它提供 AIME 每套试卷 15 道整数答案题；模型评测通常按年份和 AIME I/II 固定集合，单年两套合计 30 题。 的评测规模信息和 competition-math 方向的可复用 benchmark 坐标。","primary_link":"https://maa.org/math-competitions/american-invitational-mathematics-examination-aime","links":[],"link_count":1,"sections":9},{"id":"alignbench-chinese-alignment-2024","title":"AlignBench: Benchmarking Chinese Alignment of Large Language Models","year":2024,"venue":"ACL 2024","authors":["Xiao Liu","Xuanyu Lei","Shengyuan Wang","Yue Huang","Zhuoer Feng","Bosi Wen","Jiale Cheng","Pei Ke","Yifan Xu","Weng Lam Tam","Xiaohan Zhang","Lichao Sun","Xiaotao Gu","Hongning Wang","Jing Zhang","Minlie Huang","Yuxiao Dong","Jie Tang"],"authors_zh":"Xiao Liu、Xuanyu Lei、Shengyuan Wang、Yue Huang、Zhuoer Feng、Bosi Wen、Jiale Cheng、Pei Ke、Yifan Xu、Weng Lam Tam、Xiaohan Zhang、Lichao Sun、Xiaotao Gu、Hongning Wang、Jing Zhang、Minlie Huang、Yuxiao Dong、Jie Tang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["AlignBench combines real Chinese user scenarios, human-verified references, and rule-calibrated judging to assess multidimensional alignment.","AlignBench 结合真实中文用户场景、人工核验参考答案与规则校准评判，衡量多维对齐能力。"],"why":"AlignBench combines real Chinese user scenarios, human-verified references, and rule-calibrated judging to assess multidimensional alignment.","primary_link":"https://aclanthology.org/2024.acl-long.624/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/AlignBench"}],"link_count":3,"sections":9},{"id":"alphamath-almost-zero-2024","title":"AlphaMath Almost Zero: Process Supervision without Process","year":2024,"venue":"NeurIPS 2024","authors":["Guoxin Chen","Minpeng Liao","Chengxi Li","Kai Fan"],"authors_zh":"陈国新、廖敏鹏、李成熙、范凯","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","process_supervision","data_release","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["step_level","process_reward","trajectory_value","answer_level"],"training_use":["sft","process_supervision","reward_modeling","test_time_compute"],"construction_layer":["search_substrate","trace_writing","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["math"],"tags":["math-reasoning","mcts","search-generated-data","process-supervision","step-values","positive-negative-traces","open-release"],"status":"partial","priority":"必读","paper_type_zh":"基于搜索的过程监督和自训练配方","best_for_zh":"研究由终局答案、树搜索和执行结果构造数学推理轨迹与价值目标的数据研究者","confidence":"high","one_line":["AlphaMath converts answer-labeled GSM8K and MATH prompts into selected positive and negative MCTS paths with backed-up step-value targets.","AlphaMath 将带最终答案的 GSM8K 与 MATH 题目转化为经筛选的正负 MCTS 路径和回传步骤价值目标。"],"why":"It makes search configuration, terminal checking, failed-path retention, and value backup explicit parts of a process-supervision data recipe while exposing the limits of those constructed labels.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/30dfe47a3ccbee68cffa0c19ccb1bc00-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MARIO-Math-Reasoning/Super_MARIO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MARIO-Math-Reasoning/AlphaMath-Trainset"}],"link_count":6,"sections":9},{"id":"aya-dataset-2024","title":"Aya Dataset: An Open-Access Collection for Multilingual Instruction Tuning","year":2024,"venue":"ACL 2024 Main","authors":["Shivalika Singh","Freddie Vargus","Daniel D'souza","Börje F. Karlsson","Abinaya Mahendiran","Wei-Yin Ko","Herumb Shandilya","Jay Patel","Deividas Mataciunas","Laura O'Mahony","Mike Zhang","Ramith Hettiarachchi","Joseph Wilson","Marina Machado","Luisa Souza Moura","Dominik Krzemiński","Hakimeh Fadaei","Irem Ergün","Ifeoma Okoh","Aisha Alaagib","Oshan Mudannayake","Zaid Alyafeai","Vu Minh Chien","Sebastian Ruder","Surya Guthikonda","Emad A. Alghamdi","Sebastian Gehrmann","Niklas Muennighoff","Max Bartolo","Julia Kreutzer","Ahmet Üstün","Marzieh Fadaee","Sara Hooker"],"authors_zh":"Shivalika Singh, Freddie Vargus, Daniel D'souza, Börje F. Karlsson, Abinaya Mahendiran, Wei-Yin Ko, Herumb Shandilya, Jay Patel, Deividas Mataciunas, Laura O'Mahony, Mike Zhang, Ramith Hettiarachchi, Joseph Wilson, Marina Machado, Luisa Souza Moura, Dominik Krzemiński, Hakimeh Fadaei, Irem Ergün, Ifeoma Okoh, Aisha Alaagib, Oshan Mudannayake, Zaid Alyafeai, Vu Minh Chien, Sebastian Ruder, Surya Guthikonda, Emad A. Alghamdi, Sebastian Gehrmann, Niklas Muennighoff, Max Bartolo, Julia Kreutzer, Ahmet Üstün, Marzieh Fadaee, Sara Hooker","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","community-authored-multilingual-instructions"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"给英语占主导的指令混合补充人写多语言监督","confidence":"high","one_line":["Aya publishes 204,114 human-authored instruction-response pairs across 65 languages with task provenance.","Aya 发布 20.4114 万条覆盖 65 种语言、带任务来源的人写指令回答对。"],"why":"community-scale native multilingual instruction authorship with explicit task and language provenance","primary_link":"https://aclanthology.org/2024.acl-long.620/","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/CohereForAI/aya_dataset"},{"key":"project","label":["Project","项目主页"],"url":"https://aya.for.ai/"}],"link_count":5,"sections":9},{"id":"babilong-2024","title":"BABILong: Testing the Limits of LLMs with Long Context Reasoning-in-a-Haystack","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Yuri Kuratov","Aydar Bulatov","Petr Anokhin","Ivan Rodkin","Dmitry Sorokin","Artyom Sorokin","Mikhail Burtsev"],"authors_zh":"Yuri Kuratov 等（AIRI, Moscow, Russia、Neural Networks and Deep Learning Lab, MIPT, Dolgoprudny, Russia、London Institute for Mathematical Sciences）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["long-context-reasoning","long-context-grounding"],"tags":["benchmark","long_context_grounding","long-context-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Track 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["BABILong exposes reasoning-in-a-haystack up to very long contexts as an auditable evaluation surface.","BABILong 把超长上下文下的大海捞针式推理做成可审计的评测面。"],"why":"Chaining/counting under distractors makes it more than single-needle retrieval.","primary_link":"https://arxiv.org/abs/2406.10149","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/booydar/babilong"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/RMT-team/babilong"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/RMT-team/babilong"}],"link_count":6,"sections":9},{"id":"rest-em-2024","title":"Beyond Human Data: Scaling Self-Training for Problem-Solving with Language Models","year":2024,"venue":"TMLR 2024","authors":["Avi Singh","John D. Co-Reyes","Rishabh Agarwal","Ankesh Anand","Piyush Patil","Xavier Garcia","Peter J. Liu","James Harrison","Jaehoon Lee","Kelvin Xu","Aaron Parisi","Abhishek Kumar","Alex Alemi","Alex Rizkowsky","Azade Nova","Ben Adlam","Bernd Bohnet","Gamaleldin Elsayed","Hanie Sedghi","Igor Mordatch","Isabelle Simpson","Izzeddin Gur","Jasper Snoek","Jeffrey Pennington","Jiri Hron","Kathleen Kenealy","Kevin Swersky","Kshiteej Mahajan","Laura Culp","Lechao Xiao","Maxwell L. Bileschi","Noah Constant","Roman Novak","Rosanne Liu","Tris Warkentin","Yundi Qian","Yamini Bansal","Ethan Dyer","Behnam Neyshabur","Jascha Sohl-Dickstein","Noah Fiedel"],"authors_zh":"Avi Singh, John D. Co-Reyes, Rishabh Agarwal, Ankesh Anand, Piyush Patil, Xavier Garcia, Peter J. Liu, James Harrison, Jaehoon Lee, Kelvin Xu, Aaron Parisi, Abhishek Kumar, Alex Alemi, Alex Rizkowsky, Azade Nova, Ben Adlam, Bernd Bohnet, Gamaleldin Elsayed, Hanie Sedghi, Igor Mordatch, Isabelle Simpson, Izzeddin Gur, Jasper Snoek, Jeffrey Pennington, Jiri Hron, Kathleen Kenealy, Kevin Swersky, Kshiteej Mahajan, Laura Culp, Lechao Xiao, Maxwell L. Bileschi, Noah Constant, Roman Novak, Rosanne Liu, Tris Warkentin, Yundi Qian, Yamini Bansal, Ethan Dyer, Behnam Neyshabur, Jascha Sohl-Dickstein, Noah Fiedel","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["self_play_anchor","trace_writing"],"domains":["large-language-models","self-training-demonstrations"],"tags":["self-training-demonstrations","arxiv-2312.06585","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"经 verifier 过滤的迭代自训练","best_for_zh":"适合为数学或代码任务构建带可执行二元反馈的自训练循环的读者。","confidence":"high","one_line":["ReST-EM samples math solutions or programs from the current policy, keeps only outputs that pass binary checks, and repeats SFT to turn verified self-generation into training data.","ReST-EM 从当前 policy 采样数学解答或程序，只保留通过二元检查的输出，再重复 SFT，把经验证的自生成结果变成训练数据。"],"why":"It provides a simple iterative recipe and shows exactly where verifier quality and repeated-prompt overfitting limit self-training.","primary_link":"https://arxiv.org/abs/2312.06585","links":[],"link_count":1,"sections":9},{"id":"chartassistant-chartsft-2024","title":"ChartAssistant: A Universal Chart Multimodal Language Model via Chart-to-Table Pre-training and Multitask Instruction Tuning","year":2024,"venue":"Findings of ACL 2024","authors":["Fanqing Meng","Wenqi Shao","Quanfeng Lu","Peng Gao","Kaipeng Zhang","Yu Qiao","Ping Luo"],"authors_zh":"Fanqing Meng, Wenqi Shao, Quanfeng Lu, Peng Gao, Kaipeng Zhang, Yu Qiao, Ping Luo","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","chart-to-table-aligned-multitask-instructions"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"构建同时支持图表提取、问答和推理的统一助手","confidence":"high","one_line":["ChartAssistant couples chart-to-table alignment with a released multitask ChartSFT conversation corpus.","ChartAssistant 将图表到表格对齐与公开的多任务 ChartSFT 对话语料结合。"],"why":"using table reconstruction as an explicit bridge into a universal chart instruction mixture","primary_link":"https://aclanthology.org/2024.findings-acl.463/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenGVLab/ChartAst"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/FanqingM/ChartAssistant"}],"link_count":5,"sections":9},{"id":"charxiv-2024","title":"CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Zirui Wang","Mengzhou Xia","Luxi He","Howard Chen","Yitao Liu","Richard Zhu","Kaiqu Liang","Xindi Wu","Haotian Liu","Sadhika Malladi","Alexis Chevalier","Sanjeev Arora","Danqi Chen"],"authors_zh":"Zirui Wang 等（Princeton Language and Intelligence, Princeton University、University of Wisconsin, Madison、The University of Hong Kong）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["scientific-chart-reasoning","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","scientific-chart-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Track 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["CharXiv exposes chart understanding from arXiv papers as an auditable evaluation surface.","CharXiv 把取自 arXiv 论文的图表理解做成可审计的评测面。"],"why":"Realistic paper-chart surface with human-verified reasoning questions.","primary_link":"https://arxiv.org/abs/2406.18521","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/princeton-nlp/CharXiv"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/princeton-nlp/CharXiv"},{"key":"project","label":["Project","项目主页"],"url":"https://charxiv.github.io/"}],"link_count":6,"sections":9},{"id":"criticbench-2024","title":"CriticBench: Benchmarking LLMs for Critique-Correct Reasoning","year":2024,"venue":"Findings of ACL 2024","authors":["Zicheng Lin","Zhibin Gou","Tian Liang","Ruilin Luo","Haowei Liu","Yujiu Yang"],"authors_zh":"Zicheng Lin 等（Tsinghua University、University of Hong Kong）","tracks":["benchmarks_evaluation_surfaces","judgment_rubric_domain_expert_data"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["critique-correction","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","critique-correction"],"status":"verified","priority":"可读","paper_type_zh":"Findings of ACL 2024 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["CriticBench exposes generation-critique-correction reasoning across domains as an auditable evaluation surface.","CriticBench 把跨领域的生成、批判与纠正推理做成可审计的评测面。"],"why":"Evaluates critique as feedback surface rather than only final answers.","primary_link":"https://aclanthology.org/2024.findings-acl.91/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/CriticBench/CriticBench"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/llm-agents/CriticBench"},{"key":"project","label":["Project","项目主页"],"url":"https://criticbench.github.io/"}],"link_count":6,"sections":9},{"id":"criticeval-llm-critics-2024","title":"CriticEval: Evaluating Large-scale Language Model as Critic","year":2024,"venue":"NeurIPS 2024","authors":["Tian Lan","Wenwei Zhang","Chen Xu","Heyan Huang","Dahua Lin","Kai Chen","Xian-Ling Mao"],"authors_zh":"Tian Lan、Wenwei Zhang、Chen Xu、Heyan Huang、Dahua Lin、Kai Chen、Xian-Ling Mao","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["CriticEval evaluates scalar and textual critique across four critique dimensions and nine task settings using annotated reference critiques.","CriticEval 通过带标注参考批评的四类批评维度和九类任务，同时评测标量与文本批评。"],"why":"CriticEval evaluates scalar and textual critique across four critique dimensions and nine task settings using annotated reference critiques.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/7b7d7985f62284060d65f532ed2ea5fa-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/open-compass/CriticEval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/opencompass/CriticBench"}],"link_count":4,"sections":9},{"id":"critiquellm-informative-critique-2024","title":"CRITIQUELLM: Towards an Informative Critique Generation Model for Evaluation of Large Language Model Generation","year":2024,"venue":"ACL 2024","authors":["Pei Ke","Bosi Wen","Zhuoer Feng","Xiao Liu","Xuanyu Lei","Jiale Cheng","Shengyuan Wang","Aohan Zeng","Yuxiao Dong","Hongning Wang","Jie Tang","Minlie Huang"],"authors_zh":"Pei Ke、Bosi Wen、Zhuoer Feng、Xiao Liu、Xuanyu Lei、Jiale Cheng、Shengyuan Wang、Aohan Zeng、Yuxiao Dong、Hongning Wang、Jie Tang、Minlie Huang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["CritiqueLLM trains an informative critique generator for scalable evaluation and generation improvement.","训练能给出具体、可操作理由的批评模型，并把批评作为可扩展的生成改进反馈。"],"why":"CritiqueLLM trains an informative critique generator for scalable evaluation and generation improvement.","primary_link":"https://aclanthology.org/2024.acl-long.704/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thu-coai/CritiqueLLM"}],"link_count":3,"sections":9},{"id":"cruxeval-2024","title":"CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution","year":2024,"venue":"arXiv preprint","authors":["Alex Gu","Baptiste Rozière","Hugh Leather","Armando Solar-Lezama","Gabriel Synnaeve","Sida I. Wang"],"authors_zh":"Alex Gu 等（MIT CSAIL、Meta AI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["code-reasoning","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","code-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["CRUXEval exposes code input/output reasoning checked by execution as an auditable evaluation surface.","CRUXEval 把由执行结果核验的代码输入输出推理做成可审计的评测面。"],"why":"Targets semantic execution reasoning instead of only synthesis.","primary_link":"https://arxiv.org/abs/2401.03065","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/cruxeval-org/cruxeval"}],"link_count":3,"sections":9},{"id":"cybench-2024","title":"Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models","year":2024,"venue":"ICLR 2025 Oral / arXiv","authors":["Andy K. Zhang","Neil Perry","Riya Dulepet","Joey Ji","Celeste Menders","Justin W. Lin","Eliot Jones","Gashon Hussein","Samantha Liu","Donovan Jasper","Pura Peetathawatchai","Ari Glenn","Vikram Sivashankar","Daniel Zamoshchin","Leo Glikbarg","Derek Askaryar","Mike Yang","Teddy Zhang","Rishi Alluri","Nathan Tran","Rinnara Sangpisit","Polycarpos Yiorkadjis","Kenny Osele","Gautham Raghupathi","Dan Boneh","Daniel E. Ho","Percy Liang"],"authors_zh":"Andy K. Zhang 等（Stanford University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["cybersecurity-agents","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","cybersecurity-agents"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 Oral / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["Cybench exposes CTF/cyber tasks with flags, subtasks, and command execution as an auditable evaluation surface.","Cybench 把带 flag、子任务与命令执行的夺旗与网络安全任务做成可审计的评测面。"],"why":"Concrete executable security reasoning surface with human difficulty calibration.","primary_link":"https://arxiv.org/abs/2408.08926","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/andyzorigin/cybench"},{"key":"project","label":["Project","项目主页"],"url":"https://cybench.github.io/"}],"link_count":5,"sections":9},{"id":"dart-math-2024","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","year":2024,"venue":"NeurIPS 2024","authors":["Yuxuan Tong","Xiwen Zhang","Rui Wang","Ruidong Wu","Junxian He"],"authors_zh":"Yuxuan Tong、Xiwen Zhang、Rui Wang、Ruidong Wu、Junxian He","tracks":["data_construction_open_release_recipes"],"source_role":["data_release","construction_recipe","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","natural-language-reasoning"],"tags":["mathematics","synthetic-cot","rejection-sampling","difficulty-aware-sampling","answer-verification","supervised-fine-tuning"],"status":"partial","priority":"必读","paper_type_zh":"难度感知拒绝采样与开放数学 SFT 数据配方","best_for_zh":"研究数学推理数据构造、拒绝采样预算、答案 verifier 与 SFT 分布审计的读者","confidence":"high","one_line":["DART-Math converts MATH and corrected GSM8K training prompts into answer-verified CoT pairs by allocating more DeepSeekMath-7B-RL trials to queries with lower pass rates.","DART-Math 按 DeepSeekMath-7B-RL 的低通过率向 MATH 与修正版 GSM8K 难题分配更多采样，再用正则与 SymPy 终局答案检查筛成 CoT SFT 对；该 verifier 不验证中间推理。"],"why":"It turns a usually hidden rejection-sampling budget into an auditable data-construction choice and demonstrates that coverage and difficulty distribution can matter independently of final dataset size.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/0ef1afa0daa888d695dcd5e9513bafa3-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/dart-math"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hkust-nlp/dart-math-hard"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/hkust-nlp/dart-math"},{"key":"project","label":["Project","项目主页"],"url":"https://hkust-nlp.github.io/dart-math/"}],"link_count":7,"sections":9},{"id":"deepseek-v3-technical-report-2024","title":"DeepSeek-V3 Technical Report","year":2024,"venue":"arXiv preprint","authors":["DeepSeek-AI"],"authors_zh":"DeepSeek-AI","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["sft","distillation","evaluation"],"construction_layer":["frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","coding"],"tags":["deepseek","deepseek-v3","frontier-report","data-disclosure-ledger","moe","post-training"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型技术报告","best_for_zh":"审计模型报告的数据披露边界","confidence":"medium","one_line":["DeepSeek-V3 discloses aggregate pretraining scale and a high-level SFT/RL post-training sequence, but not the data lineage or feedback contract needed to audit the recipe.","DeepSeek-V3 披露了聚合预训练规模与高层 SFT/RL 后训练顺序，但未披露审计该配方所需的数据谱系和反馈合约。"],"why":"It makes the difference between a model technical report and a reproducible reasoning-data pipeline explicit.","primary_link":"https://arxiv.org/abs/2412.19437","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/deepseek-ai/DeepSeek-V3"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/deepseek-ai/DeepSeek-V3"}],"link_count":5,"sections":9},{"id":"dsbench-2024","title":"DSBench: How Far Are Data Science Agents from Becoming Data Science Experts?","year":2024,"venue":"ICLR 2025 poster","authors":["Liqiang Jing","Zhehui Huang","Xiaoyang Wang","Wenlin Yao","Wenhao Yu","Kaixin Ma","Hongming Zhang","Xinya Du","Dong Yu"],"authors_zh":"Liqiang Jing 等（University of Texas at Dallas、Tencent AI Lab, Seattle、University of Southern California）","tracks":["benchmarks_evaluation_surfaces","programmatically_verifiable_outcome_data"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["data-science-agents","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","data-science-agents"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 poster 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["DSBench exposes data analysis and modeling tasks over realistic files as an auditable evaluation surface.","DSBench 把基于真实文件的数据分析与建模任务做成可审计的评测面。"],"why":"Strong data-science workflow lead, source details need pass-one audit.","primary_link":"https://arxiv.org/abs/2409.07703","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LiqiangJing/DSBench"},{"key":"project","label":["Project","项目主页"],"url":"https://liqiangjing.github.io/dsbench.github.io/"}],"link_count":6,"sections":9},{"id":"dynamath-2024","title":"DynaMath: A Dynamic Visual Benchmark for Evaluating Mathematical Reasoning Robustness of Vision Language Models","year":2024,"venue":"ICLR 2025 / arXiv","authors":["Chengke Zou","Xingang Guo","Rui Yang","Junyu Zhang","Bin Hu","Huan Zhang"],"authors_zh":"Chengke Zou 等（University of Illinois at Urbana-Champaign、University of California, Berkeley）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["multimodal-math","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","multimodal-math"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["DynaMath exposes programmatically generated visual math variants as an auditable evaluation surface.","DynaMath 把程序化生成的视觉数学变体题做成可审计的评测面。"],"why":"Dynamic perturbations help test rule solving versus memorization.","primary_link":"https://arxiv.org/abs/2411.00836","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/DynaMath/DynaMath"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/DynaMath/DynaMath_Sample"},{"key":"project","label":["Project","项目主页"],"url":"https://dynamath.github.io/"}],"link_count":5,"sections":9},{"id":"realmistake-error-detection-2024","title":"Evaluating LLMs at Detecting Errors in LLM Responses","year":2024,"venue":"COLM 2024","authors":["Ryo Kamoi","Sarkar Snigdha Sarathi Das","Renze Lou","Jihyun Janice Ahn","Yilun Zhao","Xiaoxin Lu","Nan Zhang","Yusen Zhang","Ranran Haoran Zhang","Sujeeth Reddy Vummanthala","Salika Dave","Shaobo Qin","Arman Cohan","Wenpeng Yin","Rui Zhang"],"authors_zh":"Ryo Kamoi、Sarkar Snigdha Sarathi Das、Renze Lou、Jihyun Janice Ahn、Yilun Zhao、Xiaoxin Lu、Nan Zhang、Yusen Zhang、Ranran Haoran Zhang、Sujeeth Reddy Vummanthala、Salika Dave、Shaobo Qin、Arman Cohan、Wenpeng Yin、Rui Zhang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["ReaLMistake evaluates whether LLMs can detect realistic, objectively checkable errors in LLM responses.","构造真实且客观的 LLM 回答错误，揭示强模型检测错误的低召回与提示敏感性。"],"why":"ReaLMistake evaluates whether LLMs can detect realistic, objectively checkable errors in LLM responses.","primary_link":"https://openreview.net/forum?id=dnwRScljXr","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/psunlpgroup/ReaLMistake"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ryokamoi/realmistake"}],"link_count":3,"sections":9},{"id":"codeact-instruct-2024","title":"Executable Code Actions Elicit Better LLM Agents","year":2024,"venue":"ICML 2024","authors":["Xingyao Wang","Yangyi Chen","Lifan Yuan","Yizhe Zhang","Yunzhu Li","Hao Peng","Heng Ji"],"authors_zh":"Xingyao Wang, Yangyi Chen, Lifan Yuan, Yizhe Zhang, Yunzhu Li, Hao Peng, Heng Ji","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["agent-reasoning","tool-use","code","mathematics","information-seeking"],"tags":["instruction-demonstration-rationale","agent-trajectories","executable-code-actions","self-debugging","arxiv-2402.01030","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"可执行智能体示范数据集与 SFT 配方","best_for_zh":"适合用代码 action、执行反馈与自我调试轨迹训练工具型智能体的研究者。","confidence":"high","one_line":["CodeActInstruct turns five task suites into 7,139 filtered multi-turn demonstrations of executable Python actions, observations, and self-corrections for SFT.","CodeActInstruct 将五类任务改造成 7,139 条可执行 Python action、环境反馈与自我纠错的多轮示范，用于智能体监督微调。"],"why":"It makes the usually hidden interaction and recovery trace a reusable training object while exposing the task mixture and acceptance rules.","primary_link":"https://arxiv.org/abs/2402.01030","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xingyaoww/code-act"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xingyaoww/code-act"}],"link_count":4,"sections":9},{"id":"llm-factuality-survey-2024","title":"Factuality of Large Language Models: A Survey","year":2024,"venue":"EMNLP 2024","authors":["Yuxia Wang","Minghan Wang","Muhammad Arslan Manzoor","Fei Liu","Georgi Nenkov Georgiev","Rocktim Jyoti Das","Preslav Nakov"],"authors_zh":"Yuxia Wang 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["factuality","hallucination","evaluation"],"tags":["survey","factuality","hallucination","evaluation","emnlp-2024"],"status":"verified","priority":"可读","paper_type_zh":"综述","best_for_zh":"需要设计事实性评测或理解幻觉缓解结论的读者。","confidence":"high","one_line":["An EMNLP 2024 survey of why LLM answers can be factually wrong and how factuality is evaluated and improved.","系统梳理大语言模型事实性错误的成因、改进方法与开放式生成的评测难点。"],"why":"It shows why a fluent answer is not enough and why factuality checks for open-ended generation remain difficult.","primary_link":"https://aclanthology.org/2024.emnlp-main.1088/","links":[],"link_count":2,"sections":9},{"id":"ferret-grit-2024","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","year":2024,"venue":"ICLR 2024","authors":["Haoxuan You","Haotian Zhang","Zhe Gan","Xianzhi Du","Bowen Zhang","Zirui Wang","Liangliang Cao","Shih-Fu Chang","Yinfei Yang"],"authors_zh":"Haoxuan You, Haotian Zhang, Zhe Gan, Xianzhi Du, Bowen Zhang, Zirui Wang, Liangliang Cao, Shih-Fu Chang, Yinfei Yang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","multi-granularity-refer-and-ground-data"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"构建允许用户圈选任意区域并提问的视觉界面","confidence":"high","one_line":["Ferret's GRIT corpus serializes points, boxes, and free-form regions into about 1.1M grounded conversations.","Ferret 的 GRIT 语料把点、框和自由形状区域序列化为约 110 万条定位对话。"],"why":"unifying multiple region granularities in one serialized dialogue contract","primary_link":"https://iclr.cc/virtual/2024/poster/19537","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/apple/ml-ferret"},{"key":"data","label":["Data","数据"],"url":"https://github.com/apple/ml-ferret#grit-dataset"}],"link_count":4,"sections":9},{"id":"folio-2022","title":"FOLIO: Natural Language Reasoning with First-Order Logic","year":2024,"venue":"EMNLP 2024 main","authors":["Simeng Han","Hailey Schoelkopf","Yilun Zhao","Zhenting Qi","Martin Riddell","Wenfei Zhou","James Coady","David Peng","Yujie Qiao","Luke Benson","Lucy Sun","Alex Wardle-Solano","Hannah Szabó","Ekaterina Zubova","Matthew Burtell","Jonathan Fan","Yixin Liu","Brian Wong","Malcolm Sailor","Ansong Ni","Linyong Nan","Jungo Kasai","Tao Yu","Rui Zhang","Alexander R. Fabbri","Wojciech Kryściński","Semih Yavuz","Ye Liu","Xi Victoria Lin","Shafiq Joty","Yingbo Zhou","Caiming Xiong","Rex Ying","Arman Cohan","Dragomir Radev"],"authors_zh":"Simeng Han 等（Yale University、Harvard University、NVIDIA、Iowa City West High School 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["logical-reasoning","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","logical-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"EMNLP 2024 main / arXiv 的 formalizable reasoning benchmark","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["FOLIO exposes first-order logic reasoning in natural language as an auditable evaluation surface.","FOLIO 把自然语言形式的一阶逻辑推理做成可审计的评测面。"],"why":"Useful for formalizable reasoning surface and label consistency checks.","primary_link":"https://arxiv.org/abs/2209.00840","links":[{"key":"data","label":["Data","数据"],"url":"https://aclanthology.org/attachments/2024.emnlp-main.1229.data.zip"}],"link_count":4,"sections":9},{"id":"free-process-rewards-without-process-labels-2024","title":"Free Process Rewards without Process Labels","year":2024,"venue":"arXiv","authors":["Lifan Yuan","Wendi Li","Huayu Chen et al."],"authors_zh":"Yuan et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"可读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Free Process Rewards without Process Labels exposes process or trace supervision data.","提出隐式 PRM：仅用廉价的回答级标签训练 outcome reward model，即可从策略与参考模型的似然比导出过程奖励。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2412.01981","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/PRIME-RL/ImplicitPRM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Windy0822/ultrainteract_math_rollout"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/Windy0822/implicitprm"}],"link_count":4,"sections":9},{"id":"arena-hard-auto-2024","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","year":2024,"venue":"arXiv preprint","authors":["Tianle Li","Wei-Lin Chiang","Evan Frick","Lisa Dunlap","Tianhao Wu","Banghua Zhu","Joseph E. Gonzalez","Ion Stoica"],"authors_zh":"Tianle Li 等（UC Berkeley）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["chat-reasoning","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","chat-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["Arena-Hard-Auto exposes automatic pairwise judge set from hard arena prompts as an auditable evaluation surface.","Arena-Hard-Auto 把源自竞技场高难提示的自动成对评审集做成可审计的评测面。"],"why":"Harder judge-based alternative to broad chat eval.","primary_link":"https://arxiv.org/abs/2406.11939","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lmarena/arena-hard-auto"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lmarena-ai/arena-hard-auto"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/lmarena-ai/arena-hard-auto-680998796296d1462c729b6c"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/spaces/lmarena-ai/arena-hard-viewer"}],"link_count":6,"sections":9},{"id":"frontiermath-advanced-math-benchmark-2024","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","year":2024,"venue":"arXiv preprint","authors":["Elliot Glazer","Ege Erdil","Tamay Besiroglu","Diego Chicharro","Evan Chen","Alex Gunning","Caroline Falkman Olsson","Jean-Stanislas Denain","Anson Ho","Emily de Oliveira Santos","Olli Järviniemi","Matthew Barnett","Robert Sandler","Matej Vrzala","Jaime Sevilla","Qiuyu Ren","Elizabeth Pratt","Lionel Levine","Grant Barkley","Natalie Stewart","Bogdan Grechuk","Tetiana Grechuk","Shreepranav Varma Enugandla","Mark Wildon","Terence Coelho","Ahsan Z. Khan","Will Brian","Pedro Teixeira","Vinh-Kha Le","Jan Jurka","Johannes Schmitt","Sai Sanjeev Balakrishnan","Peisheng Yu","David Brodsky","Razzi Masroor","Jessica Wang","Daniel Arreola","Joseph Knight","Noah Lebowitz-Lockard","Tasos Moulinos","Josh Ducey"],"authors_zh":"Elliot Glazer 等（Epoch AI、King's College London、MIT、University of Siegen 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["advanced-mathematics","research-math","contamination-audit"],"tags":["benchmark","advanced-math","expert-written","hidden-benchmark","automated-verification"],"status":"verified","priority":"必读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"研究高级数学评测、专家题库、自动 verifier、hidden/public split 与污染控制的读者。","confidence":"medium","one_line":["FrontierMath tests frontier models on original expert-written advanced mathematics problems with controlled release and automated verification.","FrontierMath 用专家原创、低污染的高等数学题和自动验证来测试 frontier model 的真实数学推理边界。"],"why":"It turns expert problem authorship, hidden/public access, and verifier coverage into central fields for evaluating whether hard math benchmarks are reusable feedback contracts.","primary_link":"https://arxiv.org/abs/2411.04872","links":[{"key":"project","label":["Project","项目主页"],"url":"https://epoch.ai/frontiermath"}],"link_count":3,"sections":9},{"id":"llm-capabilities-survey-2024","title":"Fundamental Capabilities of Large Language Models and their Applications in Domain Scenarios: A Survey","year":2024,"venue":"ACL 2024 Long Papers","authors":["Jiawei Li","Yizhe Yang","Yu Bai","Xiaofeng Zhou","Yinghao Li","Huashan Sun","Yuhang Liu","Xingpeng Si","Yuhao Ye","Yixiao Wu","Yiguan Lin","Bin Xu","Bowen Ren","Chong Feng","Yang Gao","Heyan Huang"],"authors_zh":"Jiawei Li 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["llm-capabilities","domain-applications","evaluation"],"tags":["survey","llm-capabilities","domain-applications","evaluation","acl-2024"],"status":"verified","priority":"可读","paper_type_zh":"大语言模型基础能力综述","best_for_zh":"比较模型能力、领域需求与评测局限的读者。","confidence":"high","one_line":["An ACL survey linking fundamental LLM capabilities to their roles in domain applications.","综述大语言模型的基础能力及其在不同领域应用中的作用。"],"why":"It gives readers a way to ask which ability is needed for an application instead of treating model strength as one number.","primary_link":"https://aclanthology.org/2024.acl-long.599/","links":[],"link_count":2,"sections":9},{"id":"gorilla-apibench-2023","title":"Gorilla: Large Language Model Connected with Massive APIs","year":2024,"venue":"NeurIPS 2024 Main Conference Track / arXiv","authors":["Shishir G. Patil","Tianjun Zhang","Xin Wang","Joseph E. Gonzalez"],"authors_zh":"Shishir G. Patil、Tianjun Zhang、Xin Wang、Joseph E. Gonzalez（UC Berkeley、Microsoft Research）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["data_release","benchmark"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["state_action_level"],"training_use":["sft","agent_training","evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["tool-use","agent-environments","interactive-evaluation"],"tags":["tool-use","agent-environment","trajectory-data"],"status":"verified","priority":"必读","paper_type_zh":"NeurIPS 2024 Main Conference Track / arXiv 的 API tool-use benchmark","best_for_zh":"研究 tool-use、agent trajectory、环境反馈契约和可复现评测的读者","confidence":"high","one_line":["Gorilla / APIBench evaluates API-calling LLMs with documentation-grounded instructions and AST-based tool-call matching.","Gorilla / APIBench 用文档驱动指令和 AST 匹配评测 LLM 的 API 调用能力。"],"why":"It is a foundational API tool-use benchmark with an explicit AST-based feedback contract.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/e4c61f578ff07830f5c37378dd3ecb0d-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ShishirPatil/gorilla"},{"key":"project","label":["Project","项目主页"],"url":"https://gorilla.cs.berkeley.edu"}],"link_count":5,"sections":9},{"id":"groma-instruct-2024","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","year":2024,"venue":"ECCV 2024","authors":["Chuofan Ma","Yi Jiang","Jiannan Wu","Zehuan Yuan","Xiaojuan Qi"],"authors_zh":"Chuofan Ma, Yi Jiang, Jiannan Wu, Zehuan Yuan, Xiaojuan Qi","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","region-token-grounded-conversations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练能够讨论用户指定图像区域的多模态助手","confidence":"high","one_line":["Groma releases about 30K GPT-4V conversations that bind dialogue turns to localized visual tokens and boxes.","Groma 发布约 3 万条 GPT-4V 定位对话，把会话中的指代绑定到局部视觉 token 与边界框。"],"why":"making localized visual tokens explicit fields that can be referenced throughout a natural dialogue","primary_link":"https://doi.org/10.1007/978-3-031-72658-3_24","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/FoundationVision/Groma"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/FoundationVision/groma_instruct"}],"link_count":4,"sections":9},{"id":"hallucidoctor-2024","title":"HalluciDoctor: Mitigating Hallucinatory Toxicity in Visual Instruction Data","year":2024,"venue":"CVPR 2024","authors":["Qifan Yu","Juncheng Li","Longhui Wei","Liang Pang","Wentao Ye","Bosheng Qin","Siliang Tang","Qi Tian","Yueting Zhuang"],"authors_zh":"Qifan Yu, Juncheng Li, Longhui Wei, Liang Pang, Wentao Ye, Bosheng Qin, Siliang Tang, Qi Tian, Yueting Zhuang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","hallucination-aware-visual-data-repair"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"在训练前审计并修复视觉指令混合中的无依据回答","confidence":"high","one_line":["HalluciDoctor diagnoses and rewrites roughly 50K visual conversations whose answers contain unsupported image claims.","HalluciDoctor 诊断并改写约 5 万条含有图像无依据陈述的视觉对话。"],"why":"treating hallucination as a repairable property of training records rather than only a decoding failure","primary_link":"https://openaccess.thecvf.com/content/CVPR2024/html/Yu_HalluciDoctor_Mitigating_Hallucinatory_Toxicity_in_Visual_Instruction_Data_CVPR_2024_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Yuqifan1117/HalluciDoctor"},{"key":"data","label":["Data","数据"],"url":"https://drive.google.com/file/d/1M0dZwF6nPuZMLeAH44VhFj0RCS4KxL5D/view?usp=sharing"}],"link_count":5,"sections":9},{"id":"helpsteer-2-open-source-dataset-for-training-top-performing-reward-models-2024","title":"HelpSteer 2: Open-source dataset for training top-performing reward models","year":2024,"venue":"arXiv","authors":["Zhilin Wang","Yi Dong","Olivier Delalleau","Jiaqi Zeng","Gerald Shen","Daniel Egert","Jimmy J. Zhang","Makesh Narsimhan Sreedhar","Oleksii Kuchaiev"],"authors_zh":"Zhilin Wang、Yi Dong、Olivier Delalleau 等（NVIDIA）","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["reward_modeling","preference_learning","evaluation","audit"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"人工偏好数据、奖励模型与 SteerLM 对齐研究","best_for_zh":"需要可商用人类偏好数据和多属性奖励信号的研究者。","confidence":"high","one_line":["HelpSteer2 is a CC-BY-4.0 preference dataset of about 10,000 response pairs for efficient reward-model training.","发布 HelpSteer2 开源人类偏好数据，以可解释的多维质量属性和偏好对训练高性能通用奖励模型。"],"why":"It shows how rich human preference attributes and a small, permissively licensed pair set can support reward modeling and downstream alignment.","primary_link":"https://arxiv.org/abs/2406.08673","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA/NeMo-Aligner"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/HelpSteer2"}],"link_count":3,"sections":9},{"id":"judgebench-2024","title":"JudgeBench: A Benchmark for Evaluating LLM-based Judges","year":2024,"venue":"ICLR 2025 conference paper / arXiv","authors":["Sijun Tan","Siyuan Zhuang","Kyle Montgomery","William Y. Tang","Alejandro Cuadron","Chenguang Wang","Raluca Ada Popa","Ion Stoica"],"authors_zh":"Sijun Tan 等（UC Berkeley、Washington University in St. Louis）","tracks":["benchmarks_evaluation_surfaces","judgment_rubric_domain_expert_data","preference_reward_feedback_data"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["llm-judges","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","llm-judges"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["JudgeBench exposes objective response-pair judging for knowledge/reasoning/math/code as an auditable evaluation surface.","JudgeBench 把知识、推理、数学与代码领域的客观答案对评判做成可审计的评测面。"],"why":"Meta-evaluates LLM judges on hard correctness pairs.","primary_link":"https://arxiv.org/abs/2410.12784","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ScalerLab/JudgeBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ScalerLab/JudgeBench"}],"link_count":5,"sections":9},{"id":"knowledge-conflicts-llm-survey-2024","title":"Knowledge Conflicts for LLMs: A Survey","year":2024,"venue":"EMNLP 2024","authors":["Rongwu Xu","Zehan Qi","Zhijiang Guo","Cunxiang Wang","Hongru Wang","Yue Zhang","Wei Xu"],"authors_zh":"Rongwu Xu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["optimizer_scaffold"],"domains":["knowledge-conflicts","robustness","reasoning"],"tags":["foundations-and-primers","knowledge-conflicts","emnlp-2024","survey"],"status":"verified","priority":"可读","paper_type_zh":"知识冲突综述","best_for_zh":"关注检索增强、事实可靠性和矛盾证据处理的读者。","confidence":"high","one_line":["An EMNLP 2024 survey of conflicts between contextual and parametric knowledge in LLMs.","系统梳理大语言模型中上下文知识与参数知识发生冲突的情形。"],"why":"It makes disagreement between a prompt and a model's learned knowledge explicit and comparable.","primary_link":"https://aclanthology.org/2024.emnlp-main.486/","links":[],"link_count":2,"sections":9},{"id":"kto-prospect-theoretic-optimization-2024","title":"KTO: Model Alignment as Prospect Theoretic Optimization","year":2024,"venue":"ICML 2024","authors":["Kawin Ethayarajh","Winnie Xu","Niklas Muennighoff","Dan Jurafsky","Douwe Kiela"],"authors_zh":"Kawin Ethayarajh、Winnie Xu、Niklas Muennighoff、Dan Jurafsky、Douwe Kiela（斯坦福大学、Contextual AI）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction_following"],"tags":["kto","binary-feedback","preference-learning","alignment"],"status":"verified","priority":"必读","paper_type_zh":"二元反馈驱动的偏好优化与语言模型对齐研究","best_for_zh":"适合理解弱监督反馈如何替代成对偏好、并直接进入模型对齐训练的读者。","confidence":"high","one_line":["KTO directly aligns a policy from desirable and undesirable response labels using a prospect-theoretic objective.","KTO 将逐条回答的好/坏二元反馈直接转化为基于前景理论的对齐目标，避免要求同一提示下的成对偏好。"],"why":"It changes the minimal feedback record required by a direct alignment trainer from a pair to an independently labeled response.","primary_link":"https://arxiv.org/abs/2402.01306","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ContextualAI/HALOs"}],"link_count":4,"sections":9},{"id":"length-controlled-alpacaeval-2024","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","year":2024,"venue":"COLM 2024","authors":["Yann Dubois","Balázs Galambosi","Percy Liang","Tatsunori B. Hashimoto"],"authors_zh":"Yann Dubois、Balázs Galambosi、Percy Liang、Tatsunori B. Hashimoto（Stanford University、Independent Researcher）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["instruction-following","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","instruction-following","length_bias"],"status":"verified","priority":"必读","paper_type_zh":"COLM 2024 / arXiv 的自动评测去偏论文","best_for_zh":"关注 LLM-as-judge、偏好评测、leaderboard 去偏、评测 harness 复用边界的研究者。","confidence":"high","one_line":["Length-Controlled AlpacaEval debiases AlpacaEval-style automatic preference scores by estimating win rate at equal output length.","Length-Controlled AlpacaEval 通过在同等输出长度条件下估计胜率，降低 AlpacaEval 自动偏好评测对长回答的偏置。"],"why":"It makes a widely used LLM-as-judge leaderboard more useful for auditing judge bias rather than rewarding verbosity.","primary_link":"https://arxiv.org/abs/2404.04475","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tatsu-lab/alpaca_eval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/tatsu-lab/alpaca_eval"},{"key":"project","label":["Project","项目主页"],"url":"https://tatsu-lab.github.io/alpaca_eval/"}],"link_count":5,"sections":9},{"id":"less-data-selection-2024","title":"LESS: Selecting Influential Data for Targeted Instruction Tuning","year":2024,"venue":"ICML 2024","authors":["Mengzhou Xia","Sadhika Malladi","Suchin Gururangan","Sanjeev Arora","Danqi Chen"],"authors_zh":"Mengzhou Xia, Sadhika Malladi, Suchin Gururangan, Sanjeev Arora, Danqi Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","gradient-influence-instruction-selection"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"从大型指令池构建小规模任务专属微调子集","confidence":"high","one_line":["LESS releases gradient influence scores and task-specific instruction subsets selected from larger public pools.","LESS 公开梯度影响分数与面向具体任务筛选的指令子集。"],"why":"ranking instruction records by training-gradient influence on an explicit target task","primary_link":"https://proceedings.mlr.press/v235/xia24c.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/princeton-nlp/LESS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/princeton-nlp/less_data"}],"link_count":4,"sections":9},{"id":"livecodebench-contamination-free-code-2024","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","year":2024,"venue":"arXiv preprint","authors":["Naman Jain","King Han","Alex Gu","Wen-Ding Li","Fanjia Yan","Tianjun Zhang","Sida Wang","Armando Solar-Lezama","Koushik Sen","Ion Stoica"],"authors_zh":"Naman Jain 等（UC Berkeley、MIT、Cornell University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["contamination-resistant-code","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","contamination-resistant-code"],"status":"verified","priority":"必读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["LiveCodeBench exposes recent contest problems with execution, repair, and output prediction as an auditable evaluation surface.","LiveCodeBench 把近期竞赛题的执行、修复与输出预测做成可审计的评测面。"],"why":"Live-updated contest source makes it a top code contamination-audit candidate.","primary_link":"https://arxiv.org/abs/2403.07974","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LiveCodeBench/LiveCodeBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/livecodebench/code_generation_lite"},{"key":"project","label":["Project","项目主页"],"url":"https://livecodebench.github.io/"}],"link_count":5,"sections":9},{"id":"reviewcritique-paper-meta-reviewing-2024","title":"LLMs Assist NLP Researchers: Critique Paper (Meta-)Reviewing","year":2024,"venue":"EMNLP 2024","authors":["Jiangshu Du","Yibo Wang","Wenting Zhao","Zhongfen Deng","Shuaiqi Liu","Renze Lou","Henry Peng Zou","Pranav Narayanan Venkit","Nan Zhang","Mukund Srinath","Haoran Ranran Zhang","Vipul Gupta","Yinghui Li","Tao Li","Fei Wang","Qin Liu","Tianlin Liu","Pengzhi Gao","Congying Xia","Chen Xing","Jiayang Cheng","Zhaowei Wang","Ying Su","Raj Sanjay Shah","Ruohao Guo","Jing Gu","Haoran Li","Kangda Wei","Zihao Wang","Lu Cheng","Surangika Ranathunga","Meng Fang","Jie Fu","Fei Liu","Ruihong Huang","Eduardo Blanco","Yixin Cao","Rui Zhang","Philip S. Yu","Wenpeng Yin"],"authors_zh":"Jiangshu Du、Yibo Wang、Wenting Zhao、Zhongfen Deng、Shuaiqi Liu、Renze Lou、Henry Peng Zou、Pranav Narayanan Venkit、Nan Zhang、Mukund Srinath、Haoran Ranran Zhang、Vipul Gupta、Yinghui Li、Tao Li、Fei Wang、Qin Liu、Tianlin Liu、Pengzhi Gao、Congying Xia、Chen Xing、Jiayang Cheng、Zhaowei Wang、Ying Su、Raj Sanjay Shah、Ruohao Guo、Jing Gu、Haoran Li、Kangda Wei、Zihao Wang、Lu Cheng、Surangika Ranathunga、Meng Fang、Jie Fu、Fei Liu、Ruihong Huang、Eduardo Blanco、Yixin Cao、Rui Zhang、Philip S. Yu、Wenpeng Yin","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["ReviewCritique compares human and LLM paper reviews with expert-labeled deficient segments.","以专家标注的缺陷片段比较人类与 LLM 论文评审，并检验模型能否识别低质量评论。"],"why":"ReviewCritique compares human and LLM paper reviews with expert-labeled deficient segments.","primary_link":"https://aclanthology.org/2024.emnlp-main.292/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/jiangshdd/ReviewCritique"}],"link_count":3,"sections":9},{"id":"longalign-2024","title":"LongAlign: A Recipe for Long Context Alignment of Large Language Models","year":2024,"venue":"Findings of EMNLP 2024","authors":["Yushi Bai","Xin Lv","Jiajie Zhang","Yuze He","Ji Qi","Lei Hou","Jie Tang","Yuxiao Dong","Juanzi Li"],"authors_zh":"Yushi Bai, Xin Lv, Jiajie Zhang, Yuze He, Ji Qi, Lei Hou, Jie Tang, Yuxiao Dong, Juanzi Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","length-balanced-long-context-instructions"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"让已扩展上下文窗口的基础模型学会处理长文档指令","confidence":"high","one_line":["LongAlign-10k packages long documents, instructions, and answers with length-aware training metadata.","LongAlign-10k 将长文档、指令与回答连同长度感知训练元数据一起发布。"],"why":"treating context-length distribution and batch grouping as auditable properties of an instruction dataset","primary_link":"https://aclanthology.org/2024.findings-emnlp.74/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/LongAlign"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THUDM/LongAlign-10k"}],"link_count":5,"sections":9},{"id":"longvideobench-2024","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Haoning Wu","Dongxu Li","Bei Chen","Junnan Li"],"authors_zh":"Haoning Wu 等（官方来源未披露机构）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["long-context-video","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","long-context-video"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Track 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["LongVideoBench exposes long video-language QA with referred-context reasoning as an auditable evaluation surface.","LongVideoBench 把需要指代上下文推理的长视频语言问答做成可审计的评测面。"],"why":"Stresses temporal retrieval plus multimodal reasoning.","primary_link":"https://arxiv.org/abs/2407.15754","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/longvideobench/LongVideoBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/longvideobench/LongVideoBench"},{"key":"project","label":["Project","项目主页"],"url":"https://longvideobench.github.io/"}],"link_count":6,"sections":9},{"id":"mammoth-mathinstruct-2024","title":"MAmmoTH: Building Math Generalist Models through Hybrid Instruction Tuning","year":2024,"venue":"ICLR 2024","authors":["Xiang Yue","Xingwei Qu","Ge Zhang","Yao Fu","Wenhao Huang","Huan Sun","Yu Su","Wenhu Chen"],"authors_zh":"Xiang Yue, Xingwei Qu, Ge Zhang, Yao Fu, Wenhao Huang, Huan Sun, Yu Su, Wenhu Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["reasoning-data","mathematics","tool-use"],"tags":["instruction-demonstration-rationale","mathematical-reasoning","program-of-thought","arxiv-2309.05653","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"混合数学 rationale 数据集与 SFT 研究","best_for_zh":"适合设计自然语言与可执行程序混合数学示范语料的读者。","confidence":"high","one_line":["MAmmoTH releases MathInstruct, a 260k-record hybrid CoT/PoT math SFT mixture with newly generated and execution-filtered rationales.","MAmmoTH 公开 MathInstruct，把 13 个来源的 26 万条自然语言 CoT 与可执行 PoT 示范用于混合数学 SFT。"],"why":"It makes rationale representation and source breadth explicit data variables for math generalization.","primary_link":"https://openreview.net/forum?id=yLClGs770I","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TIGER-AI-Lab/MAmmoTH"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TIGER-Lab/MathInstruct"},{"key":"project","label":["Project","项目主页"],"url":"https://tiger-ai-lab.github.io/MAmmoTH/"}],"link_count":6,"sections":9},{"id":"math-llava-2024","title":"Math-LLaVA: Bootstrapping Mathematical Reasoning for Multimodal Large Language Models","year":2024,"venue":"Findings of EMNLP 2024","authors":["Wenhao Shi","Zhiqiang Hu","Yi Bin","Junhua Liu","Yang Yang","See-Kiong Ng","Lidong Bing","Roy Ka-Wei Lee"],"authors_zh":"Wenhao Shi, Zhiqiang Hu, Yi Bin, Junhua Liu, Yang Yang, See-Kiong Ng, Lidong Bing, Roy Ka-Wei Lee","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["reasoning-data","mathematics","multimodal"],"tags":["instruction-demonstration-data","multimodal-mathematical-reasoning","synthetic-data","arxiv-2406.17294","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"多模态数学指令数据构造研究","best_for_zh":"适合构建图像推理 SFT 数据，或审计教师生成短答案示范的读者。","confidence":"high","one_line":["Math-LLaVA turns 40k selected images into an open 360k-record multimodal mathematics instruction set and measures the contribution of selection and four synthesis operations.","Math-LLaVA 将四万张筛选图像扩展为开放的三十六万条多模态数学问答，并用消融实验区分图像选择与四类合成操作的贡献。"],"why":"It exposes a compact, inspectable recipe for increasing questions per image and provides ablations that distinguish selection from synthetic record expansion.","primary_link":"https://aclanthology.org/2024.findings-emnlp.268/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HZQ950419/Math-LLaVA"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Zhiqiang007/MathV360K"}],"link_count":6,"sections":9},{"id":"math-shepherd-verify-and-reinforce-llm-math-reasoning-2024","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","year":2024,"venue":"ACL 2024","authors":["Peiyi Wang","Lei Li","Zhihui Xie et al."],"authors_zh":"Wang et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"必读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations exposes process or trace supervision data.","提出 Math-Shepherd，用自动构造的步骤级监督训练数学过程验证器，并把逐步奖励用于候选重排和 PPO。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2312.08935","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/peiyi9979/Math-Shepherd"}],"link_count":2,"sections":9},{"id":"mathcoder-2024","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","year":2024,"venue":"ICLR 2024","authors":["Ke Wang","Houxing Ren","Aojun Zhou","Zimu Lu","Sichun Luo","Weikang Shi","Renrui Zhang","Linqi Song","Mingjie Zhan","Hongsheng Li"],"authors_zh":"Ke Wang, Houxing Ren, Aojun Zhou, Zimu Lu, Sichun Luo, Weikang Shi, Renrui Zhang, Linqi Song, Mingjie Zhan, Hongsheng Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["reasoning-data","mathematics","code"],"tags":["instruction-demonstration-rationale","mathematical-reasoning","code-execution","self-distillation","arxiv-2310.03731","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"可执行数学推理数据构造研究","best_for_zh":"适合构建工具交错数学 SFT 数据，或审计代码执行、自一致性与轨迹来源的读者。","confidence":"high","one_line":["MathCoder turns math problems into open trajectories that alternate reasoning, executable Python, execution feedback, and continued reasoning, then trains models to consume that contract.","MathCoder 将数学题转成开放轨迹，让自然语言推理、可执行 Python、执行反馈和后续推理交替出现，并用这一契约训练模型。"],"why":"It makes execution feedback a serialized post-training data object and tests which gains come from interpolated problems, live execution, and loss placement.","primary_link":"https://openreview.net/forum?id=z8TW0ttBPp","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/mathllm/MathCoder"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MathLLMs/MathCodeInstruct"}],"link_count":4,"sections":9},{"id":"mathgenie-2024","title":"MathGenie: Generating Synthetic Data with Question Back-translation for Enhancing Mathematical Reasoning of LLMs","year":2024,"venue":"ACL 2024","authors":["Zimu Lu","Aojun Zhou","Houxing Ren","Ke Wang","Weikang Shi","Junting Pan","Mingjie Zhan","Hongsheng Li"],"authors_zh":"Zimu Lu、Aojun Zhou、Houxing Ren、Ke Wang、Weikang Shi、Junting Pan、Mingjie Zhan、Hongsheng Li","tracks":["instruction_demonstration_rationale_data"],"source_role":["model_report","data_release","construction_recipe","process_supervision","scaling_study"],"verification_contract":["programmatic","environmental","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","process_supervision","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report"],"domains":["mathematical-reasoning","grade-school-math","competition-math","code-execution"],"tags":["instruction-demonstration-rationale","math-reasoning","question-back-translation","code-integrated-solutions","verification-rationales","arxiv-2402.16352","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"解答回译数学数据集与验证筛选研究","best_for_zh":"适合从解答合成数学题，或审计代码集成示范和模型生成的验证推理。","confidence":"high","one_line":["MathGenieData combines 81K code-integrated solutions, 30K verification rationales, and 170K back-translated math pairs in typed execution-aware records.","MathGenieData 汇集 81K 代码解答、30K 验证推理和 170K 解答回译题目，并保留文本、代码与执行结果。"],"why":"It makes solution-first question synthesis and rationale-bearing quality filtering independently inspectable in an openly readable math corpus.","primary_link":"https://aclanthology.org/2024.acl-long.151/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MathGenie/MathGenie"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MathGenie/MathGenieData"},{"key":"project","label":["Project","项目主页"],"url":"https://mathgenie.github.io/"}],"link_count":7,"sections":9},{"id":"simpleqa-2024","title":"Measuring short-form factuality in large language models","year":2024,"venue":"OpenAI / arXiv","authors":["Jason Wei","Nguyen Karina","Hyung Won Chung","Yunxin Joy Jiao","Spencer Papay","Amelia Glaese","John Schulman","William Fedus"],"authors_zh":"Jason Wei、Nguyen Karina、Hyung Won Chung 等（OpenAI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["factuality","short-form-qa","calibration"],"tags":["benchmark","factuality","short-form-qa","calibration"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要审计短事实问答、factuality、abstention/calibration 指标、公开 benchmark 污染风险和 judge prompt 稳定性的读者。","confidence":"high","one_line":["SimpleQA evaluates short factual answers with a public 4,326-row dataset and a three-way factuality/abstention grader.","SimpleQA 用 4,326 条短事实问答和三分类评分器评估模型的正确回答、错误回答与不作答行为。"],"why":"It is a compact factuality benchmark for separating correct answers, wrong claims, and non-attempts under a reproducible evaluator contract.","primary_link":"https://cdn.openai.com/papers/simpleqa.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/simple-evals/blob/main/simpleqa_eval.py"},{"key":"data","label":["Data","数据"],"url":"https://openaipublic.blob.core.windows.net/simple-evals/simple_qa_test_set.csv"},{"key":"project","label":["Project","项目主页"],"url":"https://openai.com/index/introducing-simpleqa/"}],"link_count":6,"sections":9},{"id":"metamathqa-2024","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","year":2024,"venue":"ICLR 2024","authors":["Longhui Yu","Weisen Jiang","Han Shi","Jincheng Yu","Zhengying Liu","Yu Zhang","James T. Kwok","Zhenguo Li","Adrian Weller","Weiyang Liu"],"authors_zh":"Longhui Yu, Weisen Jiang, Han Shi, Jincheng Yu, Zhengying Liu, Yu Zhang, James T. Kwok, Zhenguo Li, Adrian Weller, Weiyang Liu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["reasoning-data","mathematics"],"tags":["instruction-demonstration-rationale","mathematical-reasoning","question-bootstrapping","arxiv-2309.12284","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"数学问题自举与推理示范数据发布","best_for_zh":"适合追踪合成数学 SFT 数据谱系，或比较问题多样性与回答多样性的读者。","confidence":"high","one_line":["MetaMathQA bootstraps both math questions and their worked solutions into a 395k-record public SFT dataset.","MetaMathQA 把 GSM8K 与 MATH 种子题扩展成 39.5 万条答案增强、改写与正反向推理的公开 SFT 记录。"],"why":"It made question transformation, not only response sampling, a first-class axis of synthetic mathematical reasoning data.","primary_link":"https://openreview.net/forum?id=N8N0hgNDRt","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/meta-math/MetaMath"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/meta-math/MetaMathQA"},{"key":"project","label":["Project","项目主页"],"url":"https://meta-math.github.io/"}],"link_count":5,"sections":9},{"id":"mllm-as-a-judge-vision-language-2024","title":"MLLM-as-a-Judge: Assessing Multimodal LLM-as-a-Judge with Vision-Language Benchmark","year":2024,"venue":"ICML 2024 (Oral)","authors":["Dongping Chen","Ruoxi Chen","Shilin Zhang","Yaochen Wang","Yinuo Liu","Huichi Zhou","Qihui Zhang","Yao Wan","Pan Zhou","Lichao Sun"],"authors_zh":"Dongping Chen、Ruoxi Chen、Shilin Zhang、Yaochen Wang、Yinuo Liu、Huichi Zhou、Qihui Zhang、Yao Wan、Pan Zhou、Lichao Sun","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"必读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["MLLM-as-a-Judge benchmarks multimodal judges on scoring, pairwise comparison, and batch ranking.","以打分、成对比较和批量排序三类任务全面检验多模态模型担任裁判的能力。"],"why":"MLLM-as-a-Judge benchmarks multimodal judges on scoring, pairwise comparison, and batch ranking.","primary_link":"https://arxiv.org/abs/2402.04788","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/Dongping-Chen/MLLM-Judge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ONE-Lab/MLLM-as-a-Judge"},{"key":"project","label":["Project","项目主页"],"url":"https://mllm-judge.github.io/"}],"link_count":4,"sections":9},{"id":"mmc-chart-instruction-2024","title":"MMC: Advancing Multimodal Chart Understanding with Large-scale Instruction Tuning","year":2024,"venue":"NAACL 2024","authors":["Fuxiao Liu","Xiaoyang Wang","Wenlin Yao","Jianshu Chen","Kaiqiang Song","Sangwoo Cho","Yaser Yacoob","Dong Yu"],"authors_zh":"Fuxiao Liu, Xiaoyang Wang, Wenlin Yao, Jianshu Chen, Kaiqiang Song, Sangwoo Cho, Yaser Yacoob, Dong Yu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","large-scale-chart-question-answer-instructions"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练覆盖科学与日常可视化的通用图表助手","confidence":"high","one_line":["MMC packages more than 409K chart QA records and 250K alignment examples from scientific and non-scientific charts.","MMC 汇集超过 40.9 万条图表问答记录与 25 万条对齐样本，覆盖科学图表和非科学图表。"],"why":"combining scientific-chart mining with a large unified chart instruction and alignment mixture","primary_link":"https://aclanthology.org/2024.naacl-long.70/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/FuxiaoLiu/MMC"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xywang1/MMC"}],"link_count":5,"sections":9},{"id":"mmlu-pro-2024","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Spotlight / arXiv","authors":["Yubo Wang","Xueguang Ma","Ge Zhang","Yuansheng Ni","Abhranil Chandra","Shiguang Guo","Weiming Ren","Aaran Arulraj","Xuan He","Ziyan Jiang","Tianle Li","Max Ku","Kai Wang","Alex Zhuang","Rongqi Fan","Xiang Yue","Wenhu Chen"],"authors_zh":"Yubo Wang 等（University of Waterloo、University of Toronto、Carnegie Mellon University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["broad-academic","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","broad-academic"],"status":"verified","priority":"必读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Spotlight / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MMLU-Pro exposes harder cleaned MMLU-style questions with 10 answer choices as an auditable evaluation surface.","MMLU-Pro 把经清洗、含十个选项的更高难度 MMLU 风格题目做成可审计的评测面。"],"why":"Useful upgraded baseline for saturation, option-count, and quality-audit work.","primary_link":"https://arxiv.org/abs/2406.01574","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TIGER-AI-Lab/MMLU-Pro"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/TIGER-Lab/MMLU-Pro"}],"link_count":4,"sections":9},{"id":"mmmu-pro-2024","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","year":2024,"venue":"ACL 2025 / arXiv","authors":["Xiang Yue","Tianyu Zheng","Yuansheng Ni","Yubo Wang","Kai Zhang","Shengbang Tong","Yuxuan Sun","Botao Yu","Ge Zhang","Huan Sun","Yu Su","Wenhu Chen","Graham Neubig","MMMU Team"],"authors_zh":"Xiang Yue 等（Carnegie Mellon University、The Ohio State University、University of Waterloo 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["multimodal-academic-reasoning","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","multimodal-academic-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"ACL 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MMMU-Pro exposes adversarial multimodal academic QA with vision-only setting as an auditable evaluation surface.","MMMU-Pro 把含纯视觉设定的对抗性多模态学科问答做成可审计的评测面。"],"why":"Filters text-only shortcuts and is a strong MMMU successor lead.","primary_link":"https://arxiv.org/abs/2409.02813","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MMMU-Benchmark/MMMU"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MMMU/MMMU_Pro"},{"key":"project","label":["Project","项目主页"],"url":"https://mmmu-benchmark.github.io/"}],"link_count":5,"sections":9},{"id":"mol-instructions-2024","title":"Mol-Instructions: A Large-Scale Biomolecular Instruction Dataset for Large Language Models","year":2024,"venue":"ICLR 2024","authors":["Yin Fang","Xiaozhuan Liang","Ningyu Zhang","Kangwei Liu","Rui Huang","Zhuo Chen","Xiaohui Fan","Huajun Chen"],"authors_zh":"Yin Fang, Xiaozhuan Liang, Ningyu Zhang, Kangwei Liu, Rui Huang, Zhuo Chen, Xiaohui Fan, Huajun Chen","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","cross-object-biomolecular-instruction-data"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练一个模型处理分子、蛋白质和生物医学文本指令","confidence":"high","one_line":["Mol-Instructions unifies more than two million molecule, protein, and biomedical-text targets under one instruction schema.","Mol-Instructions 用统一指令结构组织超过 200 万条分子、蛋白质和生物医学文本目标。"],"why":"a shared instruction interface across symbolic biomolecular objects and natural language","primary_link":"https://iclr.cc/virtual/2024/poster/18554","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zjunlp/Mol-Instructions"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zjunlp/Mol-Instructions"}],"link_count":4,"sections":9},{"id":"mplug-docowl-15-2024","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","year":2024,"venue":"Findings of EMNLP 2024","authors":["Anwen Hu","Haiyang Xu","Jiabo Ye","Ming Yan","Liang Zhang","Bo Zhang","Chen Li","Ji Zhang","Qin Jin","Fei Huang","Jingren Zhou"],"authors_zh":"Anwen Hu, Haiyang Xu, Jiabo Ye, Ming Yan, Liang Zhang, Bo Zhang, Chen Li, Ji Zhang, Qin Jin, Fei Huang, Jingren Zhou","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","structure-aware-document-reasoning-demonstrations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练无需外部文字识别器即可回答并解释文档问题的助手","confidence":"high","one_line":["DocOwl 1.5 combines four million structure records with 25K answer-filtered document explanations.","DocOwl 1.5 把约 400 万条结构记录与 2.5 万条按答案过滤的文档解释结合起来。"],"why":"linking document structure supervision to rationale-bearing downstream conversations","primary_link":"https://aclanthology.org/2024.findings-emnlp.175/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/X-PLUG/mPLUG-DocOwl/tree/main/DocOwl1.5"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mPLUG/DocReason25K"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/X-PLUG/mPLUG-DocOwl"}],"link_count":6,"sections":9},{"id":"mumath-code-2024","title":"MuMath-Code: Combining Tool-Use Large Language Models with Multi-perspective Data Augmentation for Mathematical Reasoning","year":2024,"venue":"EMNLP 2024","authors":["Shuo Yin","Weihao You","Zhilong Ji","Guoqiang Zhong","Jinfeng Bai"],"authors_zh":"Shuo Yin、Weihao You、Zhilong Ji、Guoqiang Zhong、Jinfeng Bai","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","agent_training"],"construction_layer":["prompt_sourcing","trace_writing","search_substrate"],"domains":["mathematical-reasoning","tool-use","python-code"],"tags":["instruction-demonstration-rationale","math-reasoning","tool-use","code-interpreter","arxiv-2405.07551","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"工具集成数学推理数据集与两阶段 SFT 配方","best_for_zh":"适合训练能交替生成自然语言推理、Python 代码、执行反馈与纠错步骤的数学模型。","confidence":"high","one_line":["MuMath-Code releases 600K tool-integrated math demonstrations after a 751K natural-language stage, using answer or pseudo-answer checks to retain traces.","MuMath-Code 先用 751K 条自然语言数学数据训练，再公开 600K 条含推理、Python 执行与纠错过程的工具示范。"],"why":"It makes the transition from natural-language reasoning data to code-interpreter demonstrations explicit and ablates the value of prefix reasoning and debugging traces.","primary_link":"https://aclanthology.org/2024.emnlp-main.274/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/youweihao-tal/MuMath-Code"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/weihao1/MuMath-Code-Data"}],"link_count":6,"sections":9},{"id":"mumath-2024","title":"MuMath: Multi-perspective Data Augmentation for Mathematical Reasoning in Large Language Models","year":2024,"venue":"Findings of NAACL 2024","authors":["Weihao You","Shuo Yin","Xudong Zhao","Zhilong Ji","Guoqiang Zhong","Jinfeng Bai"],"authors_zh":"Weihao You、Shuo Yin、Xudong Zhao、Zhilong Ji、Guoqiang Zhong、Jinfeng Bai","tracks":["instruction_demonstration_rationale_data"],"source_role":["model_report","data_release","construction_recipe","scaling_study"],"verification_contract":["programmatic","judgment_required","mixed"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","distillation","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","scaling_report"],"domains":["mathematical-reasoning","grade-school-math","competition-math"],"tags":["instruction-demonstration-rationale","math-reasoning","multi-perspective-augmentation","majority-sampling","version-boundary","arxiv-2309.12284","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"多视角数学指令数据构造研究","best_for_zh":"适合增强数学监督微调数据，或比较改写、逆向构造、题目变换、嵌套规划和多数筛选。","confidence":"high","one_line":["MuMath turns GSM8K and MATH into multi-perspective natural-language SFT data, with a 304K main recipe and a distinct approximately 751K scaled release.","MuMath 以四类多视角增强把 GSM8K 和 MATH 转成自然语言训练示范，并区分 304K 主实验与约 751K 扩展发布。"],"why":"It tests whether diversifying both the problem and the visible solution structure improves tool-free math reasoning, while exposing a useful but easy-to-miss release-version boundary.","primary_link":"https://aclanthology.org/2024.findings-naacl.185/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/youweihao-tal/MuMath"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/weihao1/MuMath-Data"}],"link_count":6,"sections":9},{"id":"chain-of-thought-reasoning-survey-2024","title":"Navigate through Enigmatic Labyrinth A Survey of Chain of Thought Reasoning: Advances, Frontiers and Future","year":2024,"venue":"ACL 2024 (Long Papers)","authors":["Zheng Chu","Jingchang Chen","Qianglong Chen","Weijiang Yu","Tao He","Haotian Wang","Weihua Peng","Ming Liu","Bing Qin","Ting Liu"],"authors_zh":"Zheng Chu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["chain-of-thought","reasoning","prompting"],"tags":["foundations-and-primers","chain-of-thought","reasoning","acl-2024","survey"],"status":"verified","priority":"可读","paper_type_zh":"思维链推理综述","best_for_zh":"想在阅读具体方法前先建立思维链研究全景的读者。","confidence":"high","one_line":["An ACL 2024 survey that organizes chain-of-thought reasoning methods, current frontiers, and open questions.","系统梳理思维链推理的方法、研究前沿、挑战与开放问题。"],"why":"It helps distinguish a reasoning-chain technique from the task, evidence, and limitation with which it is reported.","primary_link":"https://aclanthology.org/2024.acl-long.65/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zchuz/CoT-Reasoning-Survey"}],"link_count":4,"sections":9},{"id":"offsetbias-debiased-evaluator-2024","title":"OffsetBias: Leveraging Debiased Data for Tuning Evaluators","year":2024,"venue":"Findings of EMNLP 2024","authors":["Junsoo Park","Seungyeon Jwa","Meiying Ren","Daeyoung Kim","Sanghyuk Choi"],"authors_zh":"Junsoo Park、Seungyeon Jwa、Meiying Ren、Daeyoung Kim、Sanghyuk Choi","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"可读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈数据审计","confidence":"","one_line":["OffsetBias releases instruction prompts, two candidate responses, and deliberately balanced evaluator labels for response comparison for bounded preference learning, reward modeling, evaluation, and audit.","OffsetBias 发布围绕其任务场景组织的候选回答比较与反馈记录，可用于有边界的偏好学习、奖励建模、评测和审计。"],"why":"The release supports feedback learning for 裁判模型的长度与位置偏差.","primary_link":"https://aclanthology.org/2024.findings-emnlp.57/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ncsoft/offsetbias"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/NCSOFT/offsetbias"}],"link_count":3,"sections":9},{"id":"olympiadbench-2024","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","year":2024,"venue":"ACL 2024 main / arXiv","authors":["Chaoqun He","Renjie Luo","Yuzhuo Bai","Shengding Hu","Zhen Leng Thai","Junhao Shen","Jinyi Hu","Xu Han","Yujie Huang","Yuxiang Zhang","Jie Liu","Lei Qi","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Chaoqun He 等（Tsinghua University、Beihang University、Wisdom Way AI Lab）","tracks":["benchmarks_evaluation_surfaces","judgment_rubric_domain_expert_data"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["math-physics-olympiad","domain-expert-benchmark"],"tags":["benchmark","domain_expert_benchmark","math-physics-olympiad"],"status":"verified","priority":"可读","paper_type_zh":"ACL 2024 main / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["OlympiadBench exposes bilingual olympiad-level math and physics, including multimodal cases as an auditable evaluation surface.","OlympiadBench 把双语奥赛级数学与物理题（含多模态题目）做成可审计的评测面。"],"why":"High-difficulty contest source with multilingual and multimodal slices.","primary_link":"https://arxiv.org/abs/2402.14008","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/OlympiadBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Hothan/OlympiadBench"},{"key":"project","label":["Project","项目主页"],"url":"https://opencompass.org.cn/leaderboard-multimodal"}],"link_count":5,"sections":9},{"id":"omni-math-2024","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark for Large Language Models","year":2024,"venue":"arXiv preprint","authors":["Bofei Gao","Feifan Song","Zhe Yang","Zefan Cai","Yibo Miao","Qingxiu Dong","Lei Li","Chenghao Ma","Liang Chen","Runxin Xu","Zhengyang Tang","Benyou Wang","Daoguang Zan","Shanghaoran Quan","Ge Zhang","Lei Sha","Yichang Zhang","Xuancheng Ren","Tianyu Liu","Baobao Chang"],"authors_zh":"Bofei Gao 等（Peking University、University of Wisconsin - Madison、Alibaba Group、Shanghai Jiao Tong University 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["olympiad-math","domain-expert-benchmark"],"tags":["benchmark","domain_expert_benchmark","olympiad-math"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["Omni-MATH exposes olympiad-level math with subdomain and difficulty labels as an auditable evaluation surface.","Omni-MATH 把带子领域与难度标注的奥赛级数学题做成可审计的评测面。"],"why":"Strong hard-math successor lead beyond MATH.","primary_link":"https://arxiv.org/abs/2410.07985","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/KbsdJames/Omni-MATH"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/KbsdJames/Omni-MATH/"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/KbsdJames/Omni-Judge"},{"key":"project","label":["Project","项目主页"],"url":"https://omni-math.github.io/"}],"link_count":6,"sections":9},{"id":"android-control-2024","title":"On the Effects of Data Scale on UI Control Agents","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track (Spotlight)","authors":["Wei Li","William Bishop","Alice Li","Chris Rawles","Folawiyo Campbell-Ajala","Divya Tyamagundlu","Oriana Riva"],"authors_zh":"Wei Li, William Bishop, Alice Li, Chris Rawles, Folawiyo Campbell-Ajala, Divya Tyamagundlu, Oriana Riva","tracks":["environment_agent_trajectory_data","data_construction_open_release_recipes"],"source_role":["data_release","benchmark","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","scaling_report","release_audit"],"domains":["agent_trajectories","mobile_ui","android","ui_control"],"tags":["android-control","mobile-ui","human-demonstrations","state-action-trajectories","imitation-learning","offline-evaluation","data-scaling","out-of-domain-generalization"],"status":"partial","priority":"可读","paper_type_zh":"移动 UI 智能体轨迹数据集与规模研究","best_for_zh":"研究移动智能体 SFT、离线动作评测、轨迹数据规模效应与环境复现审计的读者","confidence":"medium","one_line":["AndroidControl releases 15,283 human Android demonstrations with paired high-/low-level instructions and uses offline next-action matching to show that in-domain SFT scales much faster than out-of-domain high-level control.","AndroidControl 发布 15,283 条带高层目标与逐动作指令的人类 Android 示范，并以离线下一动作匹配研究数据规模；其分数不等同于在线任务成功，复用仍受许可证冲突与不可回放环境限制。"],"why":"It is both a reusable offline mobile-agent trajectory object and a warning about feedback boundaries: reported scores measure agreement with one demonstrated next action, while replay state, discarded failures, release versioning, and licensing remain incompletely specified.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/a79f3ef3b445fd4659f44648f7ea8ffd-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"data","label":["Data","数据"],"url":"https://console.cloud.google.com/storage/browser/gresearch/android_control"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/google-research/google-research/tree/master/android_control"}],"link_count":8,"sections":9},{"id":"opencodeinterpreter-2024","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","year":2024,"venue":"Findings of ACL 2024","authors":["Tianyu Zheng","Ge Zhang","Tianhao Shen","Xueling Liu","Bill Yuchen Lin","Jie Fu","Wenhu Chen","Xiang Yue"],"authors_zh":"Tianyu Zheng、Ge Zhang、Tianhao Shen、Xueling Liu、Bill Yuchen Lin、Jie Fu、Wenhu Chen、Xiang Yue","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["full_episode"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["code-generation","software-engineering","debugging","feedback-driven-refinement"],"tags":["instruction-demonstration-rationale","code-feedback","execution-feedback","iterative-refinement","arxiv-2402.14658","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"执行反馈代码示范数据集与 SFT 配方","best_for_zh":"适合训练能根据运行诊断修订程序的代码模型，或审计合成多轮反馈数据的研究者。","confidence":"high","one_line":["OpenCodeInterpreter releases 68K multi-turn code conversations and 192K turns that combine explanations, execution feedback, simulated user guidance, and revisions for SFT.","OpenCodeInterpreter 公开 68K 条多轮代码对话和 192K 个 turn，把解释、执行反馈、模拟用户指导与修订组合成可用于 SFT 的静态记录。"],"why":"It turns normally transient code-execution and feedback loops into an inspectable training object while exposing construction branches, scale, models, code, and data.","primary_link":"https://aclanthology.org/2024.findings-acl.762/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenCodeInterpreter/OpenCodeInterpreter"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/m-a-p/Code-Feedback"},{"key":"project","label":["Project","项目主页"],"url":"https://opencodeinterpreter.github.io/"}],"link_count":7,"sections":9},{"id":"openmathinstruct-1-2024","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Shubham Toshniwal","Ivan Moshkov","Sean Narenthiran","Daria Gitman","Fei Jia","Igor Gitman"],"authors_zh":"Shubham Toshniwal、Ivan Moshkov、Sean Narenthiran、Daria Gitman、Fei Jia、Igor Gitman","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["step_level"],"training_use":["sft","distillation"],"construction_layer":["trace_writing","reward_verifier_layer"],"domains":["mathematical-reasoning","grade-school-math","competition-math","code-interpreter"],"tags":["instruction-demonstration-rationale","math-reasoning","code-interpreter","open-distillation","arxiv-2402.10176","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"开放数学推理示范数据集与 SFT 配方","best_for_zh":"适合用开放模型合成数学 CoT，或审计只按最终答案验收的代码解释器轨迹。","confidence":"high","one_line":["OpenMathInstruct-1 releases 1.8M verified GSM8K/MATH code-interpreter demonstrations plus 6.6M incorrect traces under a commercially permissive license.","OpenMathInstruct-1 以开放 Mixtral teacher 生成并公开 1.8M 条经答案校验的 GSM8K/MATH 代码解释器示范，同时保留 6.6M 条错误轨迹。"],"why":"It demonstrates that prompt design, large sampling budgets, execution, and fair data selection can replace a closed teacher for math SFT.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/3d5aa9a7ce28cdc710fbd044fd3610f3-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/NVIDIA-NeMo/Skills"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nvidia/OpenMathInstruct-1"},{"key":"project","label":["Project","项目主页"],"url":"https://huggingface.co/collections/nvidia/openmath-65c5619de2ba059be0775014"}],"link_count":7,"sections":9},{"id":"judge-prompt-injection-2025","title":"Optimization-based Prompt Injection Attack to LLM-as-a-Judge","year":2024,"venue":"CCS 2024","authors":["Jiawen Shi","Zenghui Yuan","Yinuo Liu","Yue Huang","Pan Zhou","Lichao Sun","Neil Zhenqiang Gong"],"authors_zh":"Jiawen Shi, Zenghui Yuan, Yinuo Liu, Yue Huang, Pan Zhou, Lichao Sun, Neil Zhenqiang Gong","tracks":["audit_failure_contamination_verifier_attacks"],"source_role":["audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["trajectory_value"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["reasoning","evaluation","alignment"],"tags":["audit","round3"],"status":"verified","priority":"必读","paper_type_zh":"LLM 裁判的提示注入攻击与防御失效审计","best_for_zh":"部署 LLM 裁判、RLAIF、检索排序或工具选择的安全与评测团队。","confidence":"medium","one_line":["JudgeDeceiver optimizes a prompt-injection suffix that makes an LLM judge select an attacker-controlled response across response positions.","用优化得到的注入后缀劫持 LLM 裁判的候选选择，并审计换位与常见检测防御。"],"why":"It exposes that swap testing and common detection filters do not secure LLM-as-a-Judge from candidate-controlled instructions.","primary_link":"https://arxiv.org/abs/2403.17710","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ShiJiawenwen/JudgeDeceiver"}],"link_count":3,"sections":9},{"id":"orpo-monolithic-preference-2024","title":"ORPO: Monolithic Preference Optimization without Reference Model","year":2024,"venue":"EMNLP 2024","authors":["Jiwoo Hong","Noah Lee","James Thorne"],"authors_zh":"Jiwoo Hong、Noah Lee、James Thorne（KAIST AI）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction_following"],"tags":["orpo","preference-learning","sft","alignment"],"status":"verified","priority":"必读","paper_type_zh":"无参考模型的监督微调与偏好优化一体化研究","best_for_zh":"适合理解如何让偏好对直接改写监督微调目标、而不另设对齐阶段的读者。","confidence":"high","one_line":["ORPO unifies supervised fine-tuning and pairwise preference optimization through an odds-ratio penalty without a reference model.","ORPO 在监督微调中加入优选/拒选回答的赔率比惩罚，统一完成指令学习和偏好优化而不使用参考模型。"],"why":"It changes the training consumer so that a preference pair modifies SFT directly instead of being reserved for a later alignment stage.","primary_link":"https://aclanthology.org/2024.emnlp-main.626/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xfactlab/orpo"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openbmb/UltraFeedback"}],"link_count":5,"sections":9},{"id":"osprey-pixel-instruct-2024","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","year":2024,"venue":"CVPR 2024","authors":["Yuqian Yuan","Wentong Li","Jian Liu","Dongqi Tang","Xinjie Luo","Chi Qin","Lei Zhang","Jianke Zhu"],"authors_zh":"Yuqian Yuan, Wentong Li, Jian Liu, Dongqi Tang, Xinjie Luo, Chi Qin, Lei Zhang, Jianke Zhu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","pixel-grounded-visual-conversations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练能够围绕精确分割物体或部件进行解释的助手","confidence":"high","one_line":["Osprey releases 724K conversations whose targets are conditioned on arbitrary pixel masks rather than coarse boxes.","Osprey 发布 72.4 万条以任意像素掩码而非粗框为条件的视觉对话。"],"why":"making arbitrary masks first-class fields in instruction-response records","primary_link":"https://openaccess.thecvf.com/content/CVPR2024/html/Yuan_Osprey_Pixel_Understanding_with_Visual_Instruction_Tuning_CVPR_2024_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/CircleRadon/Osprey"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AntGroup-MI/Osprey-724K"}],"link_count":5,"sections":9},{"id":"phi-4-technical-report-2024","title":"Phi-4 Technical Report","year":2024,"venue":"Microsoft Research Technical Report MSR-TR-2024-57","authors":["Marah Abdin","Jyoti Aneja","Harkirat Behl","Sébastien Bubeck","Ronen Eldan","Suriya Gunasekar","Michael Harrison","Russell J. Hewett","Mojan Javaheripi","Piero Kauffmann","James R. Lee","Yin Tat Lee","Yuanzhi Li","Weishung Liu","Caio C. T. Mendes","Anh Nguyen","Eric Price","Gustavo de Rosa","Olli Saarikivi","Adil Salim","Shital Shah","Xin Wang","Rachel Ward","Yue Wu","Dingli Yu","Cyril Zhang","Yi Zhang"],"authors_zh":"Marah Abdin 等","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","release_audit"],"domains":["general_reasoning","mathematics","code","science","instruction_following","multilingual","safety"],"tags":["phi-4","microsoft","frontier-report","disclosure-ledger","synthetic-data","pivotal-token-search","sft","dpo","model-weights"],"status":"partial","priority":"必读","paper_type_zh":"前沿模型技术报告与数据披露账本","best_for_zh":"审计合成数据、SFT 与 DPO 披露边界的读者","confidence":"high","one_line":["Phi-4 releases MIT model weights and reports a 9.8T-token synthetic-heavy pipeline, 8B-token SFT, 250,297 token-local DPO examples, and about 850K judge-guided pairs, but no training corpus or stage-to-record lineage.","Phi-4 报告了以合成数据为主的预训练与后训练：约 50 类数据集、约 400B 合成 token、8B SFT token，以及由任务 oracle 辅助的 DPO；但未发布数据、完整血缘或 verifier 细节。"],"why":"It lets frontier-report readers separate report-level data disclosure, released weights, unavailable training records, mixed verifier and judge contracts, and downstream benchmark behavior instead of treating them as one evidence claim.","primary_link":"https://arxiv.org/abs/2412.08905","links":[{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/microsoft/phi-4"}],"link_count":5,"sections":9},{"id":"processbench-error-identification-2024","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","year":2024,"venue":"ACL 2025 / arXiv","authors":["Chujie Zheng","Zhenru Zhang","Beichen Zhang","Runji Lin","Keming Lu","Bowen Yu","Dayiheng Liu","Jingren Zhou","Junyang Lin"],"authors_zh":"Chujie Zheng 等（Qwen Team、Alibaba Inc.）","tracks":["benchmarks_evaluation_surfaces","preference_reward_feedback_data","process_trace_supervision_data"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["math-prm","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","math-prm"],"status":"verified","priority":"必读","paper_type_zh":"ACL 2025 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["ProcessBench exposes earliest erroneous step identification as an auditable evaluation surface.","ProcessBench 把数学推理中最早出错步骤的定位做成可审计的评测面。"],"why":"Direct PRM/critic benchmark for step-level reasoning oversight.","primary_link":"https://arxiv.org/abs/2412.06559","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/QwenLM/ProcessBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Qwen/ProcessBench"}],"link_count":5,"sections":9},{"id":"prometheus-2-an-open-source-language-model-specialized-in-evaluating-other-language-models-2024","title":"Prometheus 2: An open source language model specialized in evaluating other language models","year":2024,"venue":"arXiv","authors":["Seungone Kim","Juyoung Suk","Shayne Longpre","Bill Yuchen Lin","Jamin Shin","Sean Welleck","Graham Neubig","Moontae Lee","Kyungjae Lee","Minjoon Seo"],"authors_zh":"Seungone Kim、Juyoung Suk、Shayne Longpre、Bill Yuchen Lin、Jamin Shin、Sean Welleck、Graham Neubig、Moontae Lee、Kyungjae Lee、Minjoon Seo","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["Releases Prometheus 2, an open judge model that accepts custom rubrics and supports both absolute scoring and pairwise comparison.","发布 Prometheus 2 开放裁判模型：接收用户自定义量规，同时输出绝对评分与成对比较，替代不透明的闭源评判器。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2405.01535","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/prometheus-eval/prometheus-eval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/prometheus-eval"}],"link_count":3,"sections":9},{"id":"prometheus-vision-fine-grained-evaluation-2024","title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","year":2024,"venue":"Findings of ACL 2024","authors":["Seongyun Lee","Seungone Kim","Sue Hyun Park","Geewook Kim","Minjoon Seo"],"authors_zh":"Seongyun Lee、Seungone Kim、Sue Hyun Park、Geewook Kim、Minjoon Seo","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["Prometheus-Vision trains an open vision-language judge on 15K user-facing rubrics for image-grounded, fine-grained evaluation.","Prometheus-Vision 用 1.5 万条面向用户的量规训练开源视觉语言评判模型，进行图像依据的细粒度评测。"],"why":"Prometheus-Vision trains an open vision-language judge on 15K user-facing rubrics for image-grounded, fine-grained evaluation.","primary_link":"https://aclanthology.org/2024.findings-acl.672/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/prometheus-eval/prometheus-vision"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/prometheus-eval/Perception-Collection"}],"link_count":4,"sections":9},{"id":"prometheus-fine-grained-evaluation-2024","title":"Prometheus: Inducing Fine-Grained Evaluation Capability in Language Models","year":2024,"venue":"ICLR 2024","authors":["Seungone Kim","Jamin Shin","Yejin Cho","Joel Jang","Shayne Longpre","Hwaran Lee","Sangdoo Yun","Seongjin Shin","Sungdong Kim","James Thorne","Minjoon Seo"],"authors_zh":"Seungone Kim、Jamin Shin、Yejin Cho、Joel Jang、Shayne Longpre、Hwaran Lee、Sangdoo Yun、Seongjin Shin、Sungdong Kim、James Thorne、Minjoon Seo","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["Prometheus trains an open evaluator on GPT-4 feedback, custom rubrics, and reference answers for fine-grained scoring and critique.","Prometheus 利用 GPT-4 反馈、自定义量规和参考答案训练开源评判模型，进行细粒度评分与批评。"],"why":"Prometheus trains an open evaluator on GPT-4 feedback, custom rubrics, and reference answers for fine-grained scoring and critique.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2024/hash/803485352e61e3ebf41221e4776c9fd4-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/prometheus-eval/prometheus"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/prometheus-eval/Feedback-Collection"},{"key":"project","label":["Project","项目主页"],"url":"https://kaistai.github.io/prometheus/"}],"link_count":5,"sections":9},{"id":"puzzle-solving-reasoning-survey-2024","title":"Puzzle Solving using Reasoning of Large Language Models: A Survey","year":2024,"venue":"EMNLP 2024","authors":["Panagiotis Giadikiaroglou","Maria Lymperaiou","Giorgos Filandrianos","Giorgos Stamou"],"authors_zh":"Panagiotis Giadikiaroglou 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation","test_time_compute"],"construction_layer":["reward_verifier_layer","trace_writing"],"domains":["reasoning","puzzles","evaluation"],"tags":["foundations-and-primers","reasoning","puzzles","emnlp-2024","survey"],"status":"verified","priority":"可读","paper_type_zh":"谜题推理综述","best_for_zh":"需要选择谜题推理评测、比较提示与神经符号方法的读者。","confidence":"high","one_line":["An EMNLP 2024 survey of language-model puzzle solving, organized around rule-based and rule-less puzzles.","以规则型与非规则型谜题为主线，梳理语言模型解题的方法、数据集、基准与困难。"],"why":"It offers a concrete way to choose puzzle evaluations and distinguish a convincing-looking answer from constraint-respecting reasoning.","primary_link":"https://aclanthology.org/2024.emnlp-main.646/","links":[],"link_count":2,"sections":9},{"id":"quilt-llava-2024","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","year":2024,"venue":"CVPR 2024","authors":["Mehmet Saygin Seyfioglu","Wisdom O. Ikezogwo","Fatemeh Ghezloo","Ranjay Krishna","Linda Shapiro"],"authors_zh":"Mehmet Saygin Seyfioglu, Wisdom O. Ikezogwo, Fatemeh Ghezloo, Ranjay Krishna, Linda Shapiro","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","localized-expert-video-visual-demonstrations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练能够解释可见形态证据的病理图像助手","confidence":"high","one_line":["Quilt-LLaVA converts localized narration from open pathology videos into 107K image-question-explanation records.","Quilt-LLaVA 把开放病理视频中的局部专家讲解转换成 10.7 万条图像、问题与解释记录。"],"why":"mining expert temporal narration as localized visual demonstrations","primary_link":"https://openaccess.thecvf.com/content/CVPR2024/html/Seyfioglu_Quilt-LLaVA_Visual_Instruction_Tuning_by_Extracting_Localized_Narratives_from_Open-Source_CVPR_2024_paper.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/aldraus/quilt-llava"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/wisdomik/QUILT-LLaVA-Instruct-107K"},{"key":"project","label":["Project","项目主页"],"url":"https://quilt-llava.github.io/"}],"link_count":6,"sections":9},{"id":"ragtruth-rag-hallucination-corpus-2024","title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","year":2024,"venue":"ACL 2024","authors":["Cheng Niu","Yuanhao Wu","Juno Zhu","Siliang Xu","KaShun Shum","Randy Zhong","Juntong Song","Tong Zhang"],"authors_zh":"Cheng Niu、Yuanhao Wu、Juno Zhu、Siliang Xu、KaShun Shum、Randy Zhong、Juntong Song、Tong Zhang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["data_release","benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","sft","reward_modeling"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["factuality-grounding","evaluation"],"tags":["track7","judgment-feedback","factuality"],"status":"verified","priority":"必读","paper_type_zh":"数据集与评测论文","best_for_zh":"需要细粒度事实性、安全性或评审反馈资源的研究者。","confidence":"high","one_line":["RAGTruth is the paper's released feedback or evaluation resource.","对近 18K RAG 生成回答作实例级、词级和幻觉强度人工标注，直接对应检索证据约束下的生成事实性与细粒度错误定位。"],"why":"It makes a reusable feedback or evaluation surface available for auditing or training reasoning systems.","primary_link":"https://aclanthology.org/2024.acl-long.585/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ParticleMedia/RAGTruth"}],"link_count":3,"sections":9},{"id":"rest-mcts-2024","title":"ReST-MCTS*: LLM Self-Training via Process Reward Guided Tree Search","year":2024,"venue":"NeurIPS","authors":["Dan Zhang","Sining Zhoubian","Ziniu Hu","Yisong Yue","Yuxiao Dong","Jie Tang"],"authors_zh":"Dan Zhang, Sining Zhoubian, Ziniu Hu, Yisong Yue, Yuxiao Dong, Jie Tang","tracks":["data_construction_open_release_recipes","process_trace_supervision_data"],"source_role":["data_release","process_supervision","verifier_reward","construction_recipe","scaling_study"],"verification_contract":["programmatic","mixed"],"supervision_granularity":["answer_level","step_level","process_reward","trajectory_value"],"training_use":["sft","process_supervision","reward_modeling","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","search_substrate","scaling_report"],"domains":["math","science","reasoning"],"tags":["primary-link-checked","artifact-verified","process-reward","tree-search","self-training","math","science"],"status":"verified","priority":"必读","paper_type_zh":"基于树搜索的自训练与过程价值数据构造配方","best_for_zh":"研究 PRM 自举、搜索生成数据、轨迹筛选和迭代推理自训练的读者","confidence":"high","one_line":["ReST-MCTS* uses value-guided tree search and terminal-answer verification to create positive policy trajectories and scalar partial-solution targets for mutual self-training.","ReST-MCTS* 用价值引导树搜索与终点答案核验生成正向策略轨迹和部分解标量目标，以联合迭代训练策略模型与过程价值模型。"],"why":"For the Data Construction and Open Release Recipes track, the paper exposes a loop from prompt sourcing through tree search, terminal feedback, trajectory selection, value-label inference, and iterative reuse; its flattened public tables also make the missing lineage needed for process-level audit explicit.","primary_link":"https://arxiv.org/abs/2406.03816","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/ReST-MCTS"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zd21/ReST-MCTS-Llama3-8b-Instruct-Policy-1st"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/zd21/ReST-MCTS-Llama3-8b-Instruct-PRM-1st"},{"key":"project","label":["Project","项目主页"],"url":"https://rest-mcts.github.io/"}],"link_count":7,"sections":9},{"id":"rlaif-vs-rlhf-2024","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","year":2024,"venue":"ICML 2024","authors":["Harrison Lee","Samrat Phatale","Hassan Mansoor","Thomas Mesnard","Johan Ferret","Kellie Ren Lu","Colton Bishop","Ethan Hall","Victor Carbune","Abhinav Rastogi","Sushant Prakash"],"authors_zh":"Harrison Lee、Samrat Phatale、Hassan Mansoor、Thomas Mesnard、Johan Ferret、Kellie Lu、Colton Bishop、Ethan Hall、Victor Carbune、Abhinav Rastogi、Sushant Prakash（Google Research）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","preference_learning"],"construction_layer":["reward_verifier_layer"],"domains":["alignment","dialogue","summarization"],"tags":["rlaif","ai-feedback","reward-modeling","reinforcement-learning"],"status":"verified","priority":"必读","paper_type_zh":"AI 反馈替代人工偏好数据的强化学习对齐研究","best_for_zh":"适合需要扩展偏好标签供给、比较奖励模型与直接奖励接口的读者。","confidence":"high","one_line":["RLAIF trains alignment systems from LLM-generated feedback and compares reward-model and direct-reward variants with human-feedback RL.","RLAIF 用现成语言模型生成偏好反馈，并比较经奖励模型训练与直接获取 AI 奖励的两条强化学习路径。"],"why":"It makes the feedback labeler a scalable, replaceable training component rather than a human-only bottleneck.","primary_link":"https://proceedings.mlr.press/v235/lee24t.html","links":[],"link_count":3,"sections":9},{"id":"rlhf-v-correctional-human-feedback-2024","title":"RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback","year":2024,"venue":"CVPR 2024","authors":["Tianyu Yu","Yucheng Wang","Yue Wang","Yunlong Zhang","Ming Liu","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Tianyu Yu、Yucheng Wang、Yue Wang、Yunlong Zhang、Ming Liu、Zhiyuan Liu、Maosong Sun","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback releases image-text responses with segment-level human corrections for hallucinated content and dense preference supervision for preference learning, reward modeling, evaluation, and feedback audit.","以人工标出的视觉幻觉片段构造纠错式偏好监督，可用于密集多模态偏好优化。"],"why":"以人工标出的视觉幻觉片段构造纠错式偏好监督，可用于密集多模态偏好优化。","primary_link":"https://arxiv.org/abs/2312.00849","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RLHF-V/RLHF-V"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openbmb/RLHF-V-Dataset"}],"link_count":3,"sections":9},{"id":"sciinstruct-2024","title":"SciInstruct: a Self-Reflective Instruction Annotated Dataset for Training Scientific Language Models","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Dan Zhang","Ziniu Hu","Sining Zhoubian","Zhengxiao Du","Kaiyu Yang","Zihan Wang","Yisong Yue","Yuxiao Dong","Jie Tang"],"authors_zh":"Dan Zhang, Ziniu Hu, Sining Zhoubian, Zhengxiao Du, Kaiyu Yang, Zihan Wang, Yisong Yue, Yuxiao Dong, Jie Tang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["reasoning-data","mathematics","physics","chemistry","formal-proof"],"tags":["instruction-demonstration-rationale","scientific-reasoning","self-reflection","formal-proof","arxiv-2401.07950","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"科学推理过程数据构造与过滤研究","best_for_zh":"适合构建跨学科科学 SFT 数据，或审计自我反思教师轨迹的读者。","confidence":"high","one_line":["SciInstruct turns answer-only science questions into open worked-solution records through staged generation, self-reflection, outcome checking, and learned quality filtering.","SciInstruct 通过分阶段生成、自我反思、结果核验和质量分类，把仅有答案的科学问题转成开放的分步解答记录。"],"why":"It makes the transition from scarce answer-only scientific problems to reusable rationale SFT data explicit and evaluates both mixture components and filtering decisions.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2024/hash/02ee6b7295f720407b56c457b34c54d5-Abstract-Datasets_and_Benchmarks_Track.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/SciGLM"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zd21/SciInstruct"},{"key":"project","label":["Project","项目主页"],"url":"https://SciGLM.github.io/"}],"link_count":7,"sections":9},{"id":"sedareval-self-adaptive-rubrics-2024","title":"SedarEval: Automated Evaluation using Self-Adaptive Rubrics","year":2024,"venue":"Findings of EMNLP 2024","authors":["Zhiyuan Fan","Weinong Wang","Xing Wu","Debing Zhang"],"authors_zh":"Zhiyuan Fan、Weinong Wang、Xing Wu、Debing Zhang","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量表数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["SedarEval creates per-question adaptive rubrics with positive and penalty points and trains judges that better match human scoring.","SedarEval 为每道题生成带加分与扣分项的自适应量表，并用其训练更贴近人工评分的裁判。"],"why":"SedarEval creates per-question adaptive rubrics with positive and penalty points and trains judges that better match human scoring.","primary_link":"https://aclanthology.org/2024.findings-emnlp.984/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/wwn1233/sedareval"}],"link_count":2,"sections":9},{"id":"self-alignment-instruction-backtranslation-2024","title":"Self-Alignment with Instruction Backtranslation","year":2024,"venue":"ICLR 2024","authors":["Xian Li","Ping Yu","Chunting Zhou","Timo Schick","Omer Levy","Luke Zettlemoyer","Jason Weston","Mike Lewis"],"authors_zh":"Xian Li、Ping Yu、Chunting Zhou、Timo Schick、Omer Levy、Luke Zettlemoyer、Jason Weston、Mike Lewis（Meta AI）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing"],"domains":["alignment","instruction_following"],"tags":["instruction-backtranslation","self-alignment","synthetic-instructions","data-curation"],"status":"verified","priority":"必读","paper_type_zh":"自生成指令数据与迭代指令微调研究","best_for_zh":"适合研究如何把未标注网页文本转化为可审计指令数据的读者。","confidence":"high","one_line":["Instruction backtranslation generates and curates instructions for web documents, then uses the resulting pairs for iterative instruction tuning.","指令回译为网页文本生成并筛选相应指令，再将保留的指令—文本对用于迭代式指令微调。"],"why":"It makes the original web text the answer side of a synthetic training record and puts data generation and filtering inside the alignment loop.","primary_link":"https://proceedings.iclr.cc/paper_files/paper/2024/hash/0f8e3534eb8dee7478d4dc0e9d9a0b1a-Abstract-Conference.html","links":[],"link_count":3,"sections":9},{"id":"self-explore-2024","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","year":2024,"venue":"Findings of EMNLP 2024","authors":["Hyeonbin Hwang","Doyoung Kim","Seungone Kim","Seonghyeon Ye","Minjoon Seo"],"authors_zh":"Hyeonbin Hwang、Doyoung Kim、Seungone Kim、Seonghyeon Ye、Minjoon Seo","tracks":["data_construction_open_release_recipes"],"source_role":["construction_recipe","process_supervision","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level","pairwise_preference"],"training_use":["sft","preference_learning","process_supervision"],"construction_layer":["trace_writing","search_substrate","reward_verifier_layer","optimizer_scaffold","release_audit"],"domains":["math"],"tags":["self-training","rejection-sampling","first-error-localization","step-level-preference","dpo","mathematical-reasoning","search-generated-data","finite-search-label"],"status":"partial","priority":"可读","paper_type_zh":"拒绝采样、首个不可恢复步骤定位与细粒度偏好构造","best_for_zh":"研究如何将失败推理轨迹转化为程序化步骤监督、并审计有限搜索标签的读者","confidence":"high","one_line":["Self-Explore converts rejected math rationales into prompt/chosen/pit-step preferences by searching for the first prefix with no correct recovery among four samples; code is open, but paper-run generated pairs are not.","Self-Explore 从被拒数学推理中搜索首个四次续写均无法恢复正确答案的步骤，并构造 prompt/chosen/错误步骤偏好对；代码已开放，但论文运行生成的数据对未发布。"],"why":"It shows how answer-verifiable rejected traces can be recycled into step-local supervision while making finite-search labels, selective pair coverage, and unreleased generation ledgers central audit risks.","primary_link":"https://aclanthology.org/2024.findings-emnlp.78/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hbin0701/Self-Explore"},{"key":"data","label":["Data","数据"],"url":"https://github.com/hbin0701/Self-Explore/tree/master/data"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/hbin0701/DeepSeek_MATH_Self_Explore"}],"link_count":7,"sections":9},{"id":"spin-self-play-2024","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","year":2024,"venue":"ICML 2024","authors":["Zixiang Chen","Yihe Deng","Huizhuo Yuan","Kaixuan Ji","Quanquan Gu"],"authors_zh":"Zixiang Chen, Yihe Deng, Huizhuo Yuan, Kaixuan Ji, Quanquan Gu","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","iterative-self-play-response-data"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"在只有固定人类对话集时自举提升指令模型","confidence":"high","one_line":["SPIN releases base dialogues and iteration 0-3 policy responses that support repeated self-play fine-tuning.","SPIN 公开基础对话及第 0 至第 3 轮策略回答，用于反复自博弈微调。"],"why":"turning the model's previous responses into an explicit, versioned negative side of an iterative training dataset","primary_link":"https://proceedings.mlr.press/v235/chen24j.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/uclaml/SPIN"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/collections/UCLA-AGI/datasets-spin-65c3624e98d4b589bbc76f3a"}],"link_count":4,"sections":9},{"id":"self-rag-2024","title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","year":2024,"venue":"ICLR 2024 Oral","authors":["Akari Asai","Zeqiu Wu","Yizhong Wang","Avirup Sil","Hannaneh Hajishirzi"],"authors_zh":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avirup Sil, Hannaneh Hajishirzi","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe","model_report"],"verification_contract":["judgment_required"],"supervision_granularity":["step_level"],"training_use":["sft","distillation","test_time_compute"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["retrieval-augmented-generation","open-domain-question-answering","factuality","long-form-generation","fact-verification"],"tags":["reflection-token-data","retrieval-augmented-generation","critique-distillation","factuality","instruction-tuning"],"status":"verified","priority":"必读","paper_type_zh":"反思 token 指令数据集与检索增强生成方法","best_for_zh":"适合构建或审计检索感知、事实性与自我批判 SFT 数据的读者。","confidence":"high","one_line":["Self-RAG releases 150K instruction outputs augmented with passages and reflection tokens for adaptive retrieval-and-critique SFT.","Self-RAG 公开约 15 万条插入检索、相关性、支持度与效用 token 的指令输出，用于检索感知的自我批判式 SFT。"],"why":"It exposes a trainable record where retrieval decisions, passage relevance, claim support, and output utility are explicit tokens.","primary_link":"https://arxiv.org/abs/2310.11511","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AkariAsai/self-rag"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/selfrag/selfrag_train_data"},{"key":"project","label":["Project","项目主页"],"url":"https://selfrag.github.io/"}],"link_count":5,"sections":9},{"id":"self-rewarding-language-models-2024","title":"Self-Rewarding Language Models","year":2024,"venue":"ICML 2024","authors":["Weizhe Yuan","Richard Yuanzhe Pang","Kyunghyun Cho","Xian Li","Sainbayar Sukhbaatar","Jing Xu","Jason Weston"],"authors_zh":"Weizhe Yuan、Richard Yuanzhe Pang、Kyunghyun Cho、Xian Li、Sainbayar Sukhbaatar、Jing Xu、Jason Weston（Meta AI、纽约大学）","tracks":["training_usage_optimization_objectives","preference_reward_feedback_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling"],"construction_layer":["self_play_anchor"],"domains":["alignment","instruction_following"],"tags":["self-rewarding","iterative-dpo","llm-as-a-judge","preference-learning"],"status":"verified","priority":"必读","paper_type_zh":"自生成反馈与迭代偏好对齐研究","best_for_zh":"适合理解模型如何同时充当策略、数据生成器和评审器，并形成自迭代训练闭环的读者。","confidence":"high","one_line":["A language model generates and judges its own candidate answers, then improves through iterative DPO on the resulting preference pairs.","自奖励语言模型让同一模型生成候选回答并充当评审器，再用这些自生成的偏好对迭代进行 DPO 训练。"],"why":"It turns feedback generation from a fixed external dataset into an evolving training component that can improve together with the policy.","primary_link":"https://proceedings.mlr.press/v235/yuan24d.html","links":[],"link_count":4,"sections":9},{"id":"sharegpt4v-better-captions-2024","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","year":2024,"venue":"ECCV 2024","authors":["Lin Chen","Jinsong Li","Xiaoyi Dong","Pan Zhang","Conghui He","Jiaqi Wang","Feng Zhao","Dahua Lin"],"authors_zh":"Lin Chen, Jinsong Li, Xiaoyi Dong, Pan Zhang, Conghui He, Jiaqi Wang, Feng Zhao, Dahua Lin","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","dense-caption-supervision-for-visual-instruction-tuning"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"在视觉问答微调前构造带细节的图像落地监督","confidence":"high","one_line":["ShareGPT4V distills 100K GPT-4V captions into a captioner that expands dense visual supervision to 1.2M images.","ShareGPT4V 先收集 10 万条 GPT-4V 详尽描述，再训练描述器把高密度视觉监督扩展到 120 万张图像。"],"why":"turning caption quality into an explicit, scalable data stage before multimodal SFT","primary_link":"https://doi.org/10.1007/978-3-031-72643-9_22","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/InternLM/InternLM-XComposer/tree/main/projects/ShareGPT4V"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Lin-Chen/ShareGPT4V"},{"key":"project","label":["Project","项目主页"],"url":"https://sharegpt4v.github.io/"}],"link_count":5,"sections":9},{"id":"sharegpt4video-2024","title":"ShareGPT4Video: Improving Video Understanding and Generation with Better Captions","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks","authors":["Lin Chen","Xilin Wei","Jinsong Li","Xiaoyi Dong","Pan Zhang","Yuhang Zang","Zehui Chen","Haodong Duan","Bin Lin","Zhenyu Tang","Li Yuan","Yu Qiao","Dahua Lin","Feng Zhao","Jiaqi Wang"],"authors_zh":"Lin Chen, Xilin Wei, Jinsong Li, Xiaoyi Dong, Pan Zhang, Yuhang Zang, Zehui Chen, Haodong Duan, Bin Lin, Zhenyu Tang, Li Yuan, Yu Qiao, Dahua Lin, Feng Zhao, Jiaqi Wang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","dense-temporal-caption-demonstrations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"从公开视频构造时序推理指令记录","confidence":"high","one_line":["ShareGPT4Video releases 40K long temporal captions and a captioner for scaling video-language supervision.","ShareGPT4Video 发布 4 万条长时序描述，并提供可扩展视频语言监督的描述器。"],"why":"treating long temporally structured captions as the intermediate data engine for video instruction tuning","primary_link":"https://neurips.cc/virtual/2024/poster/97789","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ShareGPT4Omni/ShareGPT4Video"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ShareGPT4Video/ShareGPT4Video"},{"key":"project","label":["Project","项目主页"],"url":"https://sharegpt4video.github.io/"}],"link_count":6,"sections":9},{"id":"simpo-reference-free-preference-2024","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","year":2024,"venue":"NeurIPS 2024","authors":["Yu Meng","Mengzhou Xia","Danqi Chen"],"authors_zh":"Yu Meng、Mengzhou Xia、Danqi Chen（弗吉尼亚大学、普林斯顿大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","instruction_following"],"tags":["simpo","preference-learning","alignment","pairwise-data"],"status":"verified","priority":"必读","paper_type_zh":"无参考模型的直接偏好优化研究","best_for_zh":"适合理解偏好数据如何在较低显存成本下直接进入对齐训练的读者。","confidence":"high","one_line":["SimPO trains chosen/rejected preference pairs with a length-normalized implicit reward and no reference model.","SimPO 以长度归一化的平均对数概率充当隐式奖励，并用目标间隔直接训练偏好对，无需参考模型。"],"why":"It offers a lower-memory direct consumer for pairwise preference data and makes response-length handling part of the objective.","primary_link":"https://papers.neurips.cc/paper_files/paper/2024/hash/e099c1c9699814af0be873a175361713-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/princeton-nlp/SimPO"}],"link_count":4,"sections":9},{"id":"skywork-reward-bag-of-tricks-for-reward-modeling-in-llms-2024","title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","year":2024,"venue":"arXiv","authors":["Chris Yuhao Liu","Liang Zeng","Jiacai Liu","Rui Yan","Jujie He","Chaojie Wang","Shuicheng Yan","Yang Liu","Yahui Zhou"],"authors_zh":"Chris Yuhao Liu、Liang Zeng、Jiacai Liu、Rui Yan、Jujie He、Chaojie Wang、Shuicheng Yan、Yang Liu、Yahui Zhou","tracks":["preference_reward_feedback_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs addresses practical reward-modeling methods.","发布 Skywork-Reward-Preference-80K，并总结数据选择、去噪和过滤等奖励模型训练配方。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2410.18451","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SkyworkAI/Skywork-Reward"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Skywork/Skywork-Reward-Preference-80K-v0.2"}],"link_count":3,"sections":9},{"id":"spreadsheetbench-2024","title":"SpreadsheetBench: Towards Challenging Real World Spreadsheet Manipulation","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track Spotlight / arXiv","authors":["Zeyao Ma","Bohan Zhang","Jing Zhang","Jifan Yu","Xiaokang Zhang","Xiaohan Zhang","Sijia Luo","Xi Wang","Jie Tang"],"authors_zh":"Zeyao Ma 等（Renmin University of China）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["spreadsheet-agents","office-agents","file-state-tasks"],"tags":["agent_environment","trajectory_data","spreadsheet-agents","office-agents","file-state-tasks"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Spotlight 的 spreadsheet benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["SpreadsheetBench evaluates agents on challenging real-world spreadsheet manipulation.","SpreadsheetBench 用真实 Excel 操作任务和文件状态比较评测智能体。"],"why":"spreadsheet file state and manipulation tasks are central office-agent surfaces","primary_link":"https://arxiv.org/abs/2406.14991","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/RUCKBReasoning/SpreadsheetBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/KAKA22/SpreadsheetBench"},{"key":"project","label":["Project","项目主页"],"url":"https://spreadsheetbench.github.io/"}],"link_count":5,"sections":9},{"id":"step-dpo-step-wise-preference-optimization-for-long-chain-reasoning-of-llms-2024","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","year":2024,"venue":"arXiv","authors":["Xin Lai","Zhuotao Tian","Yukang Chen","Senqiao Yang","Xiangru Peng","Jiaya Jia"],"authors_zh":"Xin Lai、Zhuotao Tian、Yukang Chen、Senqiao Yang、Xiangru Peng、Jiaya Jia","tracks":["preference_reward_feedback_data"],"source_role":["process_supervision"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"可读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["Introduces Step-DPO, an algorithm that applies preference optimization to individual reasoning steps rather than only final answers.","提出 Step-DPO 步骤级偏好优化算法：把长数学推理的 DPO 监督从整题答案下沉到单个推理步骤。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2406.18629","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/dvlab-research/Step-DPO"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xinlai/Math-Step-DPO-10K"}],"link_count":3,"sections":9},{"id":"superfiltering-2024","title":"Superfiltering: Weak-to-Strong Data Filtering for Fast Instruction-Tuning","year":2024,"venue":"ACL 2024 Main","authors":["Ming Li","Yong Zhang","Shwai He","Zhitao Li","Hongyu Zhao","Jianzong Wang","Ning Cheng","Tianyi Zhou"],"authors_zh":"Ming Li, Yong Zhang, Shwai He, Zhitao Li, Hongyu Zhao, Jianzong Wang, Ning Cheng, Tianyi Zhou","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","weak-model-instruction-utility-filtering"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"在筛选与训练预算有限时挑选小规模高效指令子集","confidence":"high","one_line":["Superfiltering releases scored Alpaca-family records and tiny subsets selected by GPT-2 instruction-following difficulty.","Superfiltering 发布带 GPT-2 指令跟随难度分数的 Alpaca 系列记录及其小比例筛选子集。"],"why":"using a weak language model's conditional-loss ratio as a data-utility ranking signal","primary_link":"https://aclanthology.org/2024.acl-long.769/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/tianyi-lab/Superfiltering"},{"key":"data","label":["Data","数据"],"url":"https://github.com/tianyi-lab/Superfiltering/tree/main/data"}],"link_count":5,"sections":9},{"id":"swe-bench-multimodal-2024","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","year":2024,"venue":"ICLR 2025 / arXiv","authors":["John Yang","Carlos E. Jimenez","Alex L. Zhang","Kilian Lieret","Joyce Yang","Xindi Wu","Ori Press","Niklas Muennighoff","Gabriel Synnaeve","Karthik R. Narasimhan","Diyi Yang","Sida I. Wang","Ofir Press"],"authors_zh":"John Yang 等（Stanford University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["swe-agents","multimodal-repository-repair","visual-software"],"tags":["agent_environment","trajectory_data","swe-agents","multimodal-repository-repair","visual-software"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2025 / arXiv 的 multimodal SWE benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["SWE-bench Multimodal extends repository repair tasks to visual software domains.","SWE-bench Multimodal 将仓库修复任务扩展到视觉软件领域。"],"why":"it adds multimodal observations to repository repair environments","primary_link":"https://arxiv.org/abs/2410.03859","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-bench/SWE-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Multimodal"},{"key":"project","label":["Project","项目主页"],"url":"https://www.swebench.com/multimodal"}],"link_count":6,"sections":9},{"id":"swt-bench-real-world-bug-fix-tests-2024","title":"SWT-Bench: Testing and Validating Real-World Bug-Fixes with Code Agents","year":2024,"venue":"NeurIPS 2024 / arXiv v3 / OpenReview","authors":["Niels Mündler","Mark Niklas Müller","Jingxuan He","Martin Vechev"],"authors_zh":"Niels Mündler 等（ETH Zurich）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["swe-agents","test-generation","bug-fix-validation"],"tags":["agent_environment","trajectory_data","swe-agents","test-generation","bug-fix-validation"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 / 2025 update","best_for_zh":"关注软件工程智能体、环境轨迹数据、执行反馈和评测 harness 的研究者。","confidence":"high","one_line":["SWT-Bench tests whether code agents can generate and validate tests for real-world bug fixes.","SWT-Bench 检验代码智能体能否为真实 bug fix 生成并验证测试。"],"why":"it makes tests themselves part of the agent feedback and validation surface","primary_link":"https://arxiv.org/abs/2406.12952","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/logic-star-ai/swt-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/eth-sri/SWT-bench_bm25_27k_zsb"},{"key":"project","label":["Project","项目主页"],"url":"https://swtbench.com/"}],"link_count":6,"sections":9},{"id":"tau-bench-tool-agent-user-2024","title":"tau-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","year":2024,"venue":"arXiv preprint","authors":["Shunyu Yao","Noah Shinn","Pedram Razavi","Karthik Narasimhan"],"authors_zh":"Shunyu Yao 等（Sierra）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["optimizer_scaffold","reward_verifier_layer"],"domains":["tool-and-user-interaction","tool-api-benchmark"],"tags":["benchmark","tool_api_benchmark","tool-and-user-interaction"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["Tau-Bench exposes retail/airline user-agent-tool conversations with database-state grading as an auditable evaluation surface.","Tau-Bench 把零售与航空场景下的用户、智能体与工具对话及数据库状态判分做成可审计的评测面。"],"why":"Important state-based metric for reliability over repeated trials.","primary_link":"https://arxiv.org/abs/2406.12045","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sierra-research/tau-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://tau-bench.github.io/"}],"link_count":4,"sections":9},{"id":"berkeley-function-calling-leaderboard-2024","title":"The Berkeley Function Calling Leaderboard (BFCL): From Tool Use to Agentic Evaluation of Large Language Models","year":2024,"venue":"Official leaderboard BibTeX lists ICML 2025; no standalone paper page verified","authors":["Shishir G. Patil","Huanzhi Mao","Charlie Cheng-Jie Ji","Fanjia Yan","Vishnu Suresh","Ion Stoica","Joseph E. Gonzalez"],"authors_zh":"Shishir G. Patil 等（University of California, Berkeley、Berkeley Gorilla project）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["optimizer_scaffold","reward_verifier_layer"],"domains":["function-calling","tool-api-benchmark"],"tags":["benchmark","tool_api_benchmark","function-calling"],"status":"verified","priority":"必读","paper_type_zh":"官方 leaderboard BibTeX 标注 ICML 2025；未核验到独立论文页","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["BFCL / Berkeley Function Calling Leaderboard exposes function/API call generation with AST/execution checks as an auditable evaluation surface.","BFCL 把带抽象语法树与执行检查的函数及 API 调用生成做成可审计的评测面。"],"why":"Traceable tool-call evaluation surface beyond chat-only scoring.","primary_link":"https://gorilla.cs.berkeley.edu/leaderboard.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ShishirPatil/gorilla/tree/main/berkeley-function-call-leaderboard"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/gorilla-llm/Berkeley-Function-Calling-Leaderboard"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/spaces/gorilla-llm/berkeley-function-calling-leaderboard"}],"link_count":4,"sections":9},{"id":"llama-3-herd-2024","title":"The Llama 3 Herd of Models","year":2024,"venue":"arXiv preprint","authors":["Llama Team"],"authors_zh":"Llama Team","tracks":["frontier_reports_data_disclosure_ledger"],"source_role":["model_report","construction_recipe","process_supervision","verifier_reward","benchmark","scaling_study","audit_failure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","step_level","pairwise_preference","scalar_reward","process_reward"],"training_use":["sft","distillation","preference_learning","reward_modeling","process_supervision","evaluation","audit","safety_alignment"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","optimizer_scaffold","frontier_pipeline","scaling_report","release_audit"],"domains":["language_modeling","mathematical_reasoning","code_generation","tool_use","multilingual","long_context","instruction_following","safety_alignment"],"tags":["meta","llama-3","llama-3-1","open-weights","pretraining-data","synthetic-data","human-preferences","reward-model","rejection-sampling","supervised-finetuning","dpo","code-execution","mathematical-reasoning","process-reward","tool-use","llama-guard-3","contamination-audit","frontier-report","data-disclosure-ledger"],"status":"partial","priority":"必读","paper_type_zh":"开放权重基础模型谱系报告与数据披露账本","best_for_zh":"研究大规模预训练配比、合成后训练、DPO、reasoning/tool/safety 数据和开放权重审计的读者","confidence":"high","one_line":["Llama 3.1 discloses a 15.6T-token 50/25/17/8 pretraining mix and six rounds of reward modeling, best-of-10–30 rejection sampling, SFT, and DPO, while releasing weights and guards but not corpora, preferences, reward models, candidates, or item lineage.","Llama 3.1 披露 15.6T token、50/25/17/8 预训练配比，以及六轮 reward model、best-of-10–30 rejection sampling、SFT 与 DPO；它开放权重、推理工具和 Llama Guard 3，但不开放语料、偏好、RM、候选或 item-level lineage。"],"why":"It clearly connects aggregate corpus composition to post-training data objects and verifiers while exposing the gap between open weights and closed records, reward contracts, source rights, contamination controls, and generator-to-checkpoint lineage.","primary_link":"https://ai.meta.com/research/publications/the-llama-3-herd-of-models/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/meta-llama/llama-models"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/meta-llama/Meta-Llama-3.1-405B-Instruct"},{"key":"project","label":["Project","项目主页"],"url":"https://ai.meta.com/blog/meta-llama-3-1/"}],"link_count":6,"sections":9},{"id":"prism-alignment-dataset-subjective-multicultural-alignment-2024","title":"The PRISM Alignment Dataset: What Participatory, Representative and Individualised Human Feedback Reveals About the Subjective and Multicultural Alignment of Large Language Models","year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks Track","authors":["Hannah Rose Kirk","Alexander Whitefield","Paul Röttger","Andrew Michael Bean","Katerina Margatina","Juan Manuel Ciro","Rafael Mosquera","Max Bartolo","Adina Williams","He He","Bertie Vidgen","Scott A. Hale"],"authors_zh":"Hannah Rose Kirk、Alexander Whitefield、Paul Röttger 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","benchmark"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","pairwise_preference","full_episode"],"training_use":["preference_learning","evaluation","audit"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["multicultural-alignment","human-feedback","pluralism"],"tags":["human-feedback","multicultural","pluralism","participatory-alignment"],"status":"verified","priority":"必读","paper_type_zh":"参与式、多文化人类反馈数据集与对齐审计研究","best_for_zh":"研究偏好多元性、代表性采样、个性化对齐或人类反馈治理的读者","confidence":"high","one_line":["PRISM links 1,500 participants’ stated preferences and profiles to fine-grained ratings in 8,011 live conversations with 21 LLMs.","PRISM 将 1,500 名参与者的调查、社会人口信息与 21 个 LLM 的实时对话评分相连，揭示多文化对齐的主观差异。"],"why":"It makes the question of which humans provide alignment data empirically auditable rather than collapsing disagreement into one reward.","primary_link":"https://openreview.net/forum?id=DFr5hteojx","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HannahKirk/prism-alignment"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/HannahRoseKirk/prism-alignment"},{"key":"project","label":["Project","项目主页"],"url":"https://hannahkirk.github.io/prism-alignment/"}],"link_count":6,"sections":9},{"id":"the-agent-company-2024","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","year":2024,"venue":"NeurIPS 2025 Datasets and Benchmarks Track","authors":["Frank (Fangzheng) Xu","Yufan Song","Boxuan Li","Yuxuan Tang","Kritanjali Jain","Mengxue Bao","Zora Wang","Xuhui Zhou","Zhitong Guo","Murong Cao","Mingyang Yang","Hao Yang Lu","Amaad Martin","Zhe Su","Leander Maben","Raj Mehta","Wayne Chi","Lawrence Jang","Yiqing Xie","Shuyan Zhou","Graham Neubig"],"authors_zh":"Frank (Fangzheng) Xu, Yufan Song, Boxuan Li, Yuxuan Tang, Kritanjali Jain, Mengxue Bao, Zora Wang, Xuhui Zhou, Zhitong Guo, Murong Cao, Mingyang Yang, Hao Yang Lu, Amaad Martin, Zhe Su, Leander Maben, Raj Mehta, Wayne Chi, Lawrence Jang, Yiqing Xie, Shuyan Zhou, Graham Neubig","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["mixed"],"supervision_granularity":["state_action_level","full_episode","scalar_reward"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["software_engineering","project_management","data_science","administration","human_resources","finance","workplace_agent_interaction"],"tags":["environment-agent-trajectory-data","agent-benchmark","workplace-agent","long-horizon","tool-use","checkpoint-evaluation","simulated-colleagues","trajectory-release"],"status":"partial","priority":"必读","paper_type_zh":"工作场景智能体评测环境与评测轨迹发布","best_for_zh":"研究长程工具调用智能体、环境反馈、轨迹评测与 benchmark 审计的读者","confidence":"high","one_line":["TheAgentCompany pairs 175 simulated-company tasks with mixed checkpoint evaluators and releases model-generated results, screenshots, and full evaluation trajectories.","TheAgentCompany 将 175 个模拟公司任务、混合 checkpoint evaluator 与模型生成的结果、截图和完整评测轨迹连接起来，但轨迹日志许可、不可变环境版本、split、去污染、逐条 lineage、judge 校准和隐私同意仍未确认。"],"why":"It makes the full agent-data loop inspectable—prompt, environment state, action history, checkpoint feedback, and terminal outcome—while also exposing version drift, judge dependence, and release-rights gaps that matter before trajectories are reused.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2025/file/0d744742f6fac4d1134c019b7cef3c8a-Paper-Datasets_and_Benchmarks_Track.pdf","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/TheAgentCompany/TheAgentCompany"},{"key":"data","label":["Data","数据"],"url":"https://github.com/TheAgentCompany/experiments"},{"key":"project","label":["Project","项目主页"],"url":"https://the-agent-company.com/"}],"link_count":10,"sections":9},{"id":"tinychart-pot-2024","title":"TinyChart: Efficient Chart Understanding with Visual Token Merging and Program-of-Thoughts Learning","year":2024,"venue":"EMNLP 2024","authors":["Liang Zhang","Anwen Hu","Haiyang Xu","Ming Yan","Yichen Xu","Qin Jin","Ji Zhang","Fei Huang"],"authors_zh":"Liang Zhang, Anwen Hu, Haiyang Xu, Ming Yan, Yichen Xu, Qin Jin, Ji Zhang, Fei Huang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["sft"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["multimodal-reasoning","instruction-tuning"],"tags":["instruction-demonstration-rationale","executable-chart-program-demonstrations"],"status":"verified","priority":"必读","paper_type_zh":"开放指令数据集与构造方法研究","best_for_zh":"训练能够展示并执行数值步骤的小型图表模型","confidence":"high","one_line":["TinyChartData exposes chart questions with Python program-of-thought targets and final answers for compact chart SFT.","TinyChartData 公开图表问题、Python 思维程序与最终答案，用于紧凑图表模型的指令微调。"],"why":"making executable chart programs a released instruction target for a compact chart model","primary_link":"https://aclanthology.org/2024.emnlp-main.112/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/X-PLUG/mPLUG-DocOwl/tree/main/TinyChart"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/mPLUG/TinyChartData"}],"link_count":5,"sections":9},{"id":"score-self-correct-rl-2024","title":"Training Language Models to Self-Correct via Reinforcement Learning","year":2024,"venue":"arXiv preprint","authors":["Aviral Kumar","Vincent Zhuang","Rishabh Agarwal","Yi Su","John D Co-Reyes","Avi Singh","Kate Baumli","Shariq Iqbal","Colton Bishop","Rebecca Roelofs","Lei M Zhang","Kay McKinney","Disha Shrivastava","Cosmin Paduraru","George Tucker","Doina Precup","Feryal Behbahani","Aleksandra Faust"],"authors_zh":"Aviral Kumar、Vincent Zhuang、Rishabh Agarwal、Yi Su、John D Co-Reyes、Avi Singh、Kate Baumli、Shariq Iqbal、Colton Bishop、Rebecca Roelofs、Lei M Zhang、Kay McKinney、Disha Shrivastava、Cosmin Paduraru、George Tucker、Doina Precup、Feryal Behbahani、Aleksandra Faust","tracks":["rollout_search_test_time_trace_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","full_episode","scalar_reward"],"training_use":["rlvr","test_time_compute"],"construction_layer":["search_substrate","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","code"],"tags":["score","self-correction","online-rl","multi-turn-traces","reward-shaping","rlvr","test-time-compute","behavior-collapse"],"status":"partial","priority":"可读","paper_type_zh":"Rollout、搜索或测试时轨迹研究","best_for_zh":"需要审计推理轨迹、反馈契约、选择机制与复现边界的读者","confidence":"high","one_line":["SCoRe trains on the learner's own two-turn correction episodes, using a KL-constrained initialization followed by progress-shaped multi-turn RL with programmatic outcome rewards.","SCoRe 在学习者自身的两轮纠正轨迹上训练，先做 KL 约束的初始化，再用进度整形的多轮强化学习配合程序化结果奖励。"],"why":"It makes the collection-policy mismatch and correction-collapse problems explicit and gives this track a concrete schema for on-policy attempts, per-turn rewards, and sequential test-time traces.","primary_link":"https://arxiv.org/abs/2409.12917","links":[],"link_count":2,"sections":9},{"id":"ultrafeedback-scaled-ai-feedback-2024","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","year":2024,"venue":"ICML 2024","authors":["Ganqu Cui","Lifan Yuan","Ning Ding","Guanming Yao","Bingxiang He","Wei Zhu","Yuan Ni","Guotong Xie","Ruobing Xie","Yankai Lin","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Ganqu Cui、Lifan Yuan、Ning Ding、Guanming Yao、Bingxiang He、Wei Zhu、Yuan Ni、Guotong Xie、Ruobing Xie、Yankai Lin、Zhiyuan Liu、Maosong Sun（清华大学、伊利诺伊大学厄巴纳—香槟分校、ModelBest、平安科技、腾讯、中国人民大学、江苏语言能力协同创新中心）","tracks":["training_usage_optimization_objectives","preference_reward_feedback_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["reward_modeling","preference_learning"],"construction_layer":["reward_verifier_layer"],"domains":["alignment","instruction_following","reasoning"],"tags":["ai-feedback","reward-modeling","preference-learning","alignment"],"status":"verified","priority":"必读","paper_type_zh":"大规模 AI 反馈数据构造与对齐训练研究","best_for_zh":"适合需要构造反馈数据、训练奖励模型或搭建偏好对齐流程的研究者与工程人员。","confidence":"high","one_line":["UltraFeedback combines diverse instructions, multi-model answers, and GPT-4 critiques and scores into a large-scale alignment resource.","UltraFeedback 将多源指令、17 个模型回答与按四个维度给出的 GPT-4 反馈组织为大规模对齐资源。"],"why":"It makes feedback construction itself a reusable training interface instead of reducing every judgment to a single final label.","primary_link":"https://icml.cc/virtual/2024/poster/34726","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/UltraFeedback"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openbmb/UltraFeedback"}],"link_count":5,"sections":9},{"id":"v-star-training-verifiers-self-taught-reasoners-2024","title":"V-STaR: Training Verifiers for Self-Taught Reasoners","year":2024,"venue":"COLM 2024","authors":["Arian Hosseini","Xingdi Yuan","Nikolay Malkin","Aaron Courville","Alessandro Sordoni","Rishabh Agarwal"],"authors_zh":"Arian Hosseini、Xingdi Yuan、Nikolay Malkin、Aaron Courville、Alessandro Sordoni、Rishabh Agarwal","tracks":["preference_reward_feedback_data","data_construction_open_release_recipes"],"source_role":["construction_recipe","verifier_reward"],"verification_contract":["mixed"],"supervision_granularity":["answer_level","pairwise_preference","scalar_reward"],"training_use":["sft","preference_learning","reward_modeling","test_time_compute"],"construction_layer":["search_substrate","self_play_anchor","reward_verifier_layer","optimizer_scaffold"],"domains":["mathematics","code"],"tags":["v-star","self-improvement","rejection-sampling","synthetic-preference-data","outcome-supervision","dpo-verifier","best-of-k","verifier-refresh","gsm8k","mbpp","artifact-release-incomplete"],"status":"partial","priority":"必读","paper_type_zh":"迭代自训练与结果验证器数据配方","best_for_zh":"研究 rejection sampling、synthetic preference、outcome verifier、DPO 与 test-time compute 的读者","confidence":"medium","one_line":["V-STaR keeps correct self-generated solutions for iterative generator SFT and converts correct-versus-incorrect outcomes into DPO verifier pairs for Best-of-k ranking.","V-STaR 将正确的自生成解用于迭代式 generator SFT，并把同题正确—错误结果转换为 DPO verifier pairs，以支持 Best-of-k 排序。"],"why":"It shows how rejected generations can become reusable verifier supervision and how generator refresh changes negative examples, while making terminal-label noise, Cartesian-pair correlation, score gaming, and inference budget explicit audit objects.","primary_link":"https://arxiv.org/abs/2402.06457","links":[],"link_count":3,"sections":9},{"id":"visualwebarena-2024","title":"VisualWebArena: Evaluating Multimodal Agents on Realistic Visual Web Tasks","year":2024,"venue":"ACL 2024 / arXiv","authors":["Jing Yu Koh","Robert Lo","Lawrence Jang","Vikram Duvvur","Ming Chong Lim","Po-Yu Huang","Graham Neubig","Shuyan Zhou","Ruslan Salakhutdinov","Daniel Fried"],"authors_zh":"Jing Yu Koh 等（Carnegie Mellon University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","multimodal-web-agents"],"tags":["benchmark","agent_environment","web-browser-agents","multimodal-web-agents"],"status":"verified","priority":"暂缓","paper_type_zh":"ACL 2024 的 multimodal web-agent benchmark","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"high","one_line":["VisualWebArena turns visually grounded website tasks into an auditable browser-agent evaluation surface.","VisualWebArena 将视觉网页任务变成可审计的浏览器智能体评测表面。"],"why":"it adds visual UI grounding to realistic browser-agent evaluation","primary_link":"https://arxiv.org/abs/2401.13649","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/web-arena-x/visualwebarena"},{"key":"data","label":["Data","数据"],"url":"https://github.com/web-arena-x/visualwebarena/tree/main/config_files"},{"key":"project","label":["Project","项目主页"],"url":"https://jykoh.com/vwa"}],"link_count":6,"sections":9},{"id":"vlfeedback-ai-feedback-dataset-vlm-alignment-2024","title":"VLFeedback: A Large-Scale AI Feedback Dataset for Large Vision-Language Models Alignment","year":2024,"venue":"EMNLP 2024","authors":["Lei Li","Zhihui Xie","Mukai Li","Shunian Chen","Peiyi Wang","Liang Chen","Yazheng Yang","Benyou Wang","Lingpeng Kong","Qi Liu"],"authors_zh":"Lei Li、Zhihui Xie、Mukai Li 等","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward","pairwise_preference"],"training_use":["reward_modeling","preference_learning","evaluation"],"construction_layer":["prompt_sourcing","trace_writing","reward_verifier_layer","release_audit"],"domains":["vision-language","alignment","safety"],"tags":["feedback-data","vision-language","reward-modeling","ai-feedback"],"status":"verified","priority":"可读","paper_type_zh":"大规模多模态 AI 反馈数据集与 VLM 对齐研究","best_for_zh":"研究视觉语言奖励模型、DPO 或 AI 反馈审计的读者","confidence":"high","one_line":["VLFeedback releases GPT-4V-scored multimodal responses across helpfulness, visual faithfulness, and ethical considerations for VLM preference training.","VLFeedback 用 GPT-4V 按有帮助性、视觉忠实性和伦理考量为多模态回答打分并生成理由，支持 VLM 偏好训练。"],"why":"It exposes a scalable multimodal feedback contract while making judge agreement and failure modes inspectable.","primary_link":"https://aclanthology.org/2024.emnlp-main.358/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/vlf-silkie/VLFeedback"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MMInstruction/VLFeedback"},{"key":"project","label":["Project","项目主页"],"url":"https://vlf-silkie.github.io/"}],"link_count":5,"sections":9},{"id":"weblinx-2024","title":"WebLINX: Real-World Website Navigation with Multi-Turn Dialogue","year":2024,"venue":"ICML 2024 Spotlight / PMLR 235","authors":["Xing Han Lù","Zdeněk Kasner","Siva Reddy"],"authors_zh":"Xing Han Lù 等（Mila / McGill / Charles University）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit","agent_training"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","multi-turn-dialogue","web-trajectories"],"tags":["benchmark","agent_environment","web-browser-agents","multi-turn-dialogue","web-trajectories"],"status":"verified","priority":"可读","paper_type_zh":"ICML 2024 / arXiv 的 web-navigation dataset","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"high","one_line":["WebLINX packages multi-turn dialogue web navigation demonstrations for agent training and evaluation.","WebLINX 将多轮对话式网页导航示范整理为 Web 智能体数据。"],"why":"it adds dialogue-conditioned real website navigation traces","primary_link":"https://arxiv.org/abs/2402.05930","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/McGill-NLP/weblinx"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/McGill-NLP/WebLINX"},{"key":"project","label":["Project","项目主页"],"url":"https://mcgill-nlp.github.io/weblinx/"}],"link_count":6,"sections":9},{"id":"webvoyager-2024","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","year":2024,"venue":"ACL 2024 main / arXiv","authors":["Hongliang He","Wenlin Yao","Kaixin Ma","Wenhao Yu","Yong Dai","Hongming Zhang","Zhenzhong Lan","Dong Yu"],"authors_zh":"Hongliang He 等（Zhejiang University、Tencent AI Lab、Westlake University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["environmental","mixed"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit","agent_training"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","multimodal-web-agents"],"tags":["benchmark","agent_environment","web-browser-agents","multimodal-web-agents"],"status":"verified","priority":"可读","paper_type_zh":"ACL 2024 main / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"high","one_line":["WebVoyager evaluates multimodal agents on live website browsing tasks.","WebVoyager 在真实网页浏览任务上评测多模态智能体。"],"why":"it is a widely used live-web multimodal agent benchmark","primary_link":"https://arxiv.org/abs/2401.13919","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MinorJerry/WebVoyager"},{"key":"data","label":["Data","数据"],"url":"https://github.com/MinorJerry/WebVoyager/tree/main/data"}],"link_count":5,"sections":9},{"id":"wildguard-open-one-stop-moderation-tools-for-safety-risks-jailbreaks-and-refusals-of-llms-2024","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","year":2024,"venue":"arXiv","authors":["Seungju Han","Kavel Rao","Allyson Ettinger","Liwei Jiang","Bill Yuchen Lin","Nathan Lambert","Yejin Choi","Nouha Dziri"],"authors_zh":"Seungju Han、Kavel Rao、Allyson Ettinger、Liwei Jiang、Bill Yuchen Lin、Nathan Lambert、Yejin Choi、Nouha Dziri","tracks":["preference_reward_feedback_data"],"source_role":["data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"必读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs addresses open moderation tools for safety risks and jailbreaks.","发布 WildGuardMix 安全审核数据和 WildGuard 模型，用统一标签识别有害意图、回答风险、拒答与越狱攻击。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2406.18495","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/wildguard"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/wildguardmix"}],"link_count":3,"sections":9},{"id":"wizardlm-2024","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","year":2024,"venue":"ICLR 2024","authors":["Can Xu","Qingfeng Sun","Kai Zheng","Xiubo Geng","Pu Zhao","Jiazhan Feng","Chongyang Tao","Qingwei Lin","Daxin Jiang"],"authors_zh":"Can Xu, Qingfeng Sun, Kai Zheng, Xiubo Geng, Pu Zhao, Jiazhan Feng, Chongyang Tao, Qingwei Lin, Daxin Jiang","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing"],"domains":["large-language-models","synthetic-instruction-generation"],"tags":["synthetic-instruction-generation","arxiv-2304.12244","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"演化式指令—回答数据构造","best_for_zh":"适合把简单种子扩展成更难 SFT 任务、并要求来源链可审计的研究者。","confidence":"high","one_line":["WizardLM turns 52K seed instructions into a 250K candidate pool through explicit depth and breadth operations before teacher answering.","WizardLM 用显式深度和广度操作把 52K 条种子指令扩展成 250K 候选，再生成教师回答。"],"why":"It makes the intended increase in instruction difficulty traceable before the response becomes SFT supervision.","primary_link":"https://arxiv.org/abs/2304.12244","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/nlpxucan/WizardLM"}],"link_count":2,"sections":9},{"id":"workarena-plusplus-2024","title":"WorkArena++: Towards Compositional Planning and Reasoning-based Common Knowledge Work Tasks","year":2024,"venue":"arXiv; official WorkArena GitHub labels NeurIPS 2024","authors":["Léo Boisvert","Megh Thakkar","Maxime Gasse","Massimo Caccia","Thibault Le Sellier De Chezelles","Quentin Cappart","Nicolas Chapados","Alexandre Lacoste","Alexandre Drouin"],"authors_zh":"Léo Boisvert 等（ServiceNow Research）","tracks":["environment_agent_trajectory_data"],"source_role":["benchmark","agent_environment","data_release"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit","agent_training"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","enterprise-workflows","compositional-planning"],"tags":["benchmark","agent_environment","web-browser-agents","enterprise-workflows","compositional-planning"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2024 Datasets and Benchmarks Track","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"high","one_line":["WorkArena++ extends enterprise web-agent evaluation toward compositional planning trajectories.","WorkArena++ 将企业 Web 智能体评测推进到组合规划轨迹。"],"why":"it makes compositional enterprise web tasks and trajectories explicit","primary_link":"https://arxiv.org/abs/2407.05291","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ServiceNow/WorkArena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ServiceNow/WorkArena-Instances"},{"key":"project","label":["Project","项目主页"],"url":"https://servicenow.github.io/WorkArena/"}],"link_count":5,"sections":9},{"id":"workarena-2024","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","year":2024,"venue":"ICML 2024 / PMLR 235 / arXiv","authors":["Alexandre Drouin","Maxime Gasse","Massimo Caccia","Issam H. Laradji","Manuel Del Verme","Tom Marty","Léo Boisvert","Megh Thakkar","Quentin Cappart","David Vazquez","Nicolas Chapados","Alexandre Lacoste"],"authors_zh":"Alexandre Drouin 等（ServiceNow Research / Mila 等）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["web-browser-agents","enterprise-workflows"],"tags":["benchmark","agent_environment","web-browser-agents","enterprise-workflows"],"status":"verified","priority":"可读","paper_type_zh":"ICML 2024","best_for_zh":"关注 Web/浏览器智能体、环境化评测、轨迹数据、验证器和污染风险的研究者。","confidence":"high","one_line":["WorkArena evaluates web agents on stateful enterprise knowledge-work tasks.","WorkArena 用有状态企业知识工作任务评测 Web 智能体。"],"why":"it exposes realistic enterprise SaaS tasks with stateful browser execution","primary_link":"https://arxiv.org/abs/2403.07718","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/ServiceNow/WorkArena"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ServiceNow/WorkArena-Instances"},{"key":"project","label":["Project","项目主页"],"url":"https://servicenow.github.io/WorkArena/"}],"link_count":6,"sections":9},{"id":"infinitebench-2024","title":"∞Bench: Extending Long Context Evaluation Beyond 100K Tokens","year":2024,"venue":"ACL 2024 Long Papers / arXiv","authors":["Xinrong Zhang","Yingfa Chen","Shengding Hu","Zihang Xu","Junhao Chen","Moo Khai Hao","Xu Han","Zhen Leng Thai","Shuo Wang","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Xinrong Zhang 等（Department of Computer Science and Technology, Tsinghua University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["long-context","long-context-grounding"],"tags":["benchmark","long_context_grounding","long-context"],"status":"verified","priority":"可读","paper_type_zh":"ACL 2024 Long Papers / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["InfiniteBench exposes 100K+ token tasks across retrieval, code, math, novels, dialogue as an auditable evaluation surface.","InfiniteBench 把检索、代码、数学、小说与对话上超过十万词元的长上下文任务做成可审计的评测面。"],"why":"Stresses ultra-long dependencies and retrieval-plus-reasoning.","primary_link":"https://arxiv.org/abs/2402.13718","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/InfiniteBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xinrongzhang2022/InfiniteBench"}],"link_count":5,"sections":9},{"id":"mathematical-reasoning-survey-2023","title":"A Survey of Deep Learning for Mathematical Reasoning","year":2023,"venue":"ACL 2023 (Long Papers)","authors":["Pan Lu","Liang Qiu","Wenhao Yu","Sean Welleck","Kai-Wei Chang"],"authors_zh":"Pan Lu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["mathematical-reasoning","benchmarks","theorem-proving"],"tags":["foundations-and-primers","mathematical-reasoning","benchmarks","acl-2023","survey"],"status":"verified","priority":"可读","paper_type_zh":"数学推理综述","best_for_zh":"需要把数学题、证明任务、数据和评测方式对应起来的读者。","confidence":"high","one_line":["An ACL 2023 survey of mathematical-reasoning tasks, datasets, methods, benchmarks, and future directions.","梳理数学推理的任务、数据集、方法、基准与未来问题。"],"why":"It turns a broad area into concrete choices about problem types, data, evaluation, and model methods.","primary_link":"https://aclanthology.org/2023.acl-long.817/","links":[],"link_count":2,"sections":9},{"id":"agentbench-2023","title":"AgentBench: Evaluating LLMs as Agents","year":2023,"venue":"ICLR 2024","authors":["Xiao Liu","Hao Yu","Hanchen Zhang","Yifan Xu","Xuanyu Lei","Hanyu Lai","Yu Gu","Hangliang Ding","Kaiwen Men","Kejuan Yang","Shudan Zhang","Xiang Deng","Aohan Zeng","Zhengxiao Du","Chenhui Zhang","Sheng Shen","Tianjun Zhang","Yu Su","Huan Sun","Minlie Huang","Yuxiao Dong","Jie Tang"],"authors_zh":"Xiao Liu 等（Tsinghua University、The Ohio State University、UC Berkeley）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["tool-use","agent-environments","interactive-evaluation"],"tags":["tool-use","agent-environment","trajectory-data"],"status":"verified","priority":"必读","paper_type_zh":"ICLR 2024 的 benchmark / evaluation surface","best_for_zh":"研究 tool-use、agent trajectory、环境反馈契约和可复现评测的读者","confidence":"medium","one_line":["AgentBench contributes a tool-use or agent-environment surface with an explicit evaluation or verification contract.","AgentBench 提供带明确评测契约的工具使用与智能体环境评测面，用各环境特有的成功与失败信号量化推理、决策和指令遵循行为。"],"why":"It makes the action schema, environment state, and success predicate visible enough for reasoning-data curation and audit.","primary_link":"https://arxiv.org/abs/2308.03688","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/AgentBench"}],"link_count":3,"sections":9},{"id":"llava-rlhf-factually-augmented-2023","title":"Aligning Large Multimodal Models with Factually Augmented RLHF","year":2023,"venue":"ACL 2024 Findings","authors":["Zhuqing Jiang","Yue Zhang"],"authors_zh":"Zhuqing Jiang、Yue Zhang","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈审计研究者","confidence":"","one_line":["Aligning Large Multimodal Models with Factually Augmented RLHF releases image-text conversations with human preferences over answers and factually augmented reward-learning context for preference learning, reward modeling, evaluation, and feedback audit.","图文问答的人类偏好比较结合图像事实信息，可用于降低多模态奖励模型的幻觉。"],"why":"图文问答的人类偏好比较结合图像事实信息，可用于降低多模态奖励模型的幻觉。","primary_link":"https://aclanthology.org/2024.findings-acl.775/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/llava-rlhf/LLaVA-RLHF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/zhiqings/LLaVA-Human-Preference-10K"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/zhiqings/LLaVA-RLHF-Data"}],"link_count":5,"sections":9},{"id":"ares-rag-evaluation-2023","title":"ARES: An Automated Evaluation Framework for Retrieval-Augmented Generation Systems","year":2023,"venue":"NAACL 2024","authors":["Jon Saad-Falcon","Omar Khattab","Christopher Potts","Matei Zaharia"],"authors_zh":"Jon Saad-Falcon 等（Stanford University、Databricks、UC Berkeley）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["rag-evaluation","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","rag-evaluation"],"status":"verified","priority":"可读","paper_type_zh":"NAACL 2024 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["ARES exposes automated RAG evaluation with synthetic training and PPI as an auditable evaluation surface.","ARES 把带合成训练数据与预测驱动推断的自动化检索增强生成评测做成可审计的评测面。"],"why":"Useful RAG evaluator lead with statistical validation angle.","primary_link":"https://arxiv.org/abs/2311.09476","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/stanford-futuredata/ARES"},{"key":"data","label":["Data","数据"],"url":"https://github.com/stanford-futuredata/ARES/tree/main/datasets"},{"key":"project","label":["Project","项目主页"],"url":"https://ares-ai.vercel.app/"}],"link_count":5,"sections":9},{"id":"c-eval-2023","title":"C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models","year":2023,"venue":"NeurIPS 2023","authors":["Yuzhen Huang","Yuzhuo Bai","Zhihao Zhu","Junlei Zhang","Jinghan Zhang","Tangjun Su","Junteng Liu","Chuancheng Lv","Yikai Zhang","Jiayi Lei","Yao Fu","Maosong Sun","Junxian He"],"authors_zh":"Yuzhen Huang 等（Shanghai Jiao Tong University、Tsinghua University、University of Edinburgh、The Hong Kong University of Science and Technology）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["chinese-academic-exams","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","chinese-academic-exams"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2023 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["C-Eval exposes Chinese multi-level/multi-subject exam QA as an auditable evaluation surface.","C-Eval 将 Chinese multi-level/multi-subject exam QA 做成可审计的评测面。"],"why":"Important multilingual/domain comparison surface.","primary_link":"https://arxiv.org/abs/2305.08322","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SJTU-LIT/ceval"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/ceval/ceval-exam"},{"key":"project","label":["Project","项目主页"],"url":"https://cevalbenchmark.com/"}],"link_count":5,"sections":9},{"id":"tulu-2-2023","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","year":2023,"venue":"arXiv","authors":["Hamish Ivison","Yizhong Wang","Valentina Pyatkin","Nathan Lambert","Matthew Peters","Pradeep Dasigi","Joel Jang","David Wadden","Noah A. Smith","Iz Beltagy","Hannaneh Hajishirzi"],"authors_zh":"Hamish Ivison, Yizhong Wang, Valentina Pyatkin, Nathan Lambert, Matthew Peters, Pradeep Dasigi, Joel Jang, David Wadden, Noah A. Smith, Iz Beltagy, Hannaneh Hajishirzi","tracks":["foundations_and_primers"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","pairwise_preference"],"training_use":["sft","preference_learning","evaluation"],"construction_layer":["trace_writing","release_audit"],"domains":["large-language-models","practical-alignment-and-data-quality-primers"],"tags":["practical-alignment-and-data-quality-primers","arxiv-2311.10702","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"开放指令微调与偏好优化配方","best_for_zh":"适合从 SFT 数据、DPO 反馈、checkpoint 和评测四个层面构建或审计开放助手的研究者。","confidence":"high","one_line":["Tulu 2 connects a 326K instruction mixture, scalable DPO, open checkpoints, and common evaluation in one reproducible adaptation suite.","Tulu 2 把 32.6 万条指令混合数据、可扩展 DPO、开放 checkpoint 和统一评测连接成可复现的适配套件。"],"why":"It exposes which instruction records and preference pairs drive each stage of an open assistant recipe.","primary_link":"https://arxiv.org/abs/2311.10702","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/open-instruct"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/allenai/tulu-v2-sft-mixture"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/collections/allenai/tulu-v2-suite-6551b56e743e6349aab45101"}],"link_count":4,"sections":9},{"id":"bird-text-to-sql-2023","title":"Can LLM Already Serve as A Database Interface? A BIg Bench for Large-Scale Database Grounded Text-to-SQLs","year":2023,"venue":"NeurIPS 2023 Datasets and Benchmarks / arXiv","authors":["Jinyang Li","Binyuan Hui","Ge Qu","Jiaxi Yang","Binhua Li","Bowen Li","Bailin Wang","Bowen Qin","Rongyu Cao","Ruiying Geng","Nan Huo","Xuanhe Zhou","Chenhao Ma","Guoliang Li","Kevin Chang","Fei Huang","Reynold Cheng","Yongbin Li"],"authors_zh":"Jinyang Li 等（The University of Hong Kong, Alibaba Group）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["text-to-sql","database-grounded-reasoning","large-database-benchmark"],"tags":["benchmark","text-to-sql","database-grounded-reasoning","large-database-benchmark"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要大数据库 text-to-SQL、execution accuracy 和 valid-efficiency scorer 审计线索的读者。","confidence":"high","one_line":["BIRD expands text-to-SQL toward larger databases, external knowledge, and more realistic database-grounded reasoning.","BIRD 用 12,751 个 text-to-SQL pairs、95 个大数据库和约 33.4GB 内容扩展 database-grounded 评测。"],"why":"It supplies a reusable evaluation coordinate for text-to-sql, database-grounded-reasoning work.","primary_link":"https://arxiv.org/abs/2305.03111","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/bird"},{"key":"project","label":["Project","项目主页"],"url":"https://bird-bench.github.io/"}],"link_count":4,"sections":9},{"id":"cmmlu-2023","title":"CMMLU: Measuring massive multitask language understanding in Chinese","year":2023,"venue":"arXiv preprint","authors":["Haonan Li","Yixuan Zhang","Fajri Koto","Yifei Yang","Hai Zhao","Yeyun Gong","Nan Duan","Timothy Baldwin"],"authors_zh":"Haonan Li 等（MBZUAI、LibrAI、Shanghai Jiao Tong University、Microsoft Research Asia 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["chinese-academic-and-cultural-knowledge","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","chinese-academic-and-cultural-knowledge"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["CMMLU exposes Chinese multitask language understanding questions as an auditable evaluation surface.","CMMLU 把中文多任务语言理解题做成可审计的评测面。"],"why":"Useful for multilingual benchmark coverage and possible overlap analysis with C-Eval.","primary_link":"https://arxiv.org/abs/2306.09212","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/haonan-li/CMMLU"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/haonan-li/cmmlu"}],"link_count":4,"sections":9},{"id":"dpo-direct-preference-optimization-2023","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","year":2023,"venue":"NeurIPS 2023","authors":["Rafael Rafailov","Archit Sharma","Eric Mitchell","Christopher D. Manning","Stefano Ermon","Chelsea Finn"],"authors_zh":"Rafael Rafailov、Archit Sharma、Eric Mitchell、Christopher D. Manning、Stefano Ermon、Chelsea Finn（斯坦福大学）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","dialogue","summarization"],"tags":["dpo","preference-learning","alignment","pairwise-data"],"status":"verified","priority":"必读","paper_type_zh":"直接偏好优化与成对偏好数据训练研究","best_for_zh":"适合需要理解偏好对如何直接进入模型对齐训练的读者。","confidence":"high","one_line":["DPO directly optimizes a policy from chosen/rejected preference pairs using a reference-policy likelihood ratio.","DPO 用参考策略的似然比直接从优选/拒选回答对优化模型，无需单独训练奖励模型。"],"why":"It is the standard training consumer against which later preference-data selection and construction work is usually evaluated.","primary_link":"https://proceedings.neurips.cc/paper_files/paper/2023/hash/a85b405ed65c6477a4fe8302b5e06ce7-Abstract-Conference.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/eric-mitchell/direct-preference-optimization"}],"link_count":4,"sections":9},{"id":"ultrachat-2023","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","year":2023,"venue":"arXiv","authors":["Ning Ding","Yulin Chen","Bokai Xu","Yujia Qin","Zhi Zheng","Shengding Hu","Zhiyuan Liu","Maosong Sun","Bowen Zhou"],"authors_zh":"Ning Ding, Yulin Chen, Bokai Xu, Yujia Qin, Zhi Zheng, Shengding Hu, Zhiyuan Liu, Maosong Sun, Bowen Zhou","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing"],"domains":["large-language-models","synthetic-instruction-generation"],"tags":["synthetic-instruction-generation","arxiv-2305.14233","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"合成多轮对话数据构造","best_for_zh":"适合生成或审计大规模多轮聊天 SFT 语料的研究者。","confidence":"high","one_line":["UltraChat generates both sides of 1.47M topic-steered conversations, turning explicit user-demand sectors into multi-turn SFT data.","UltraChat 同时生成 147 万段主题引导对话的双方发言，把明确的用户需求类别转成多轮 SFT 数据。"],"why":"It makes synthetic user-turn provenance and conversational context visible in a large chat-training corpus.","primary_link":"https://arxiv.org/abs/2305.14233","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thunlp/UltraChat"}],"link_count":2,"sections":9},{"id":"gaokao-bench-2023","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","year":2023,"venue":"arXiv preprint","authors":["Xiaotian Zhang","Chunyang Li","Yi Zong","Zhengyu Ying","Liang He","Xipeng Qiu","Tianxiang Sun","Peng Li","Shiqiao Meng","Yanjun Zheng","Jun Zhan","Zhangyue Yin","Xiannian Hu","Guofeng Quan"],"authors_zh":"Xiaotian Zhang 等（School of Computer Science, Fudan University、School of Computer Science and Technology, East China Normal University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["chinese-gaokao-exams","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","chinese-gaokao-exams"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 Chinese exam benchmark","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["GAOKAO-Bench evaluates LLMs on Chinese Gaokao objective and subjective exam questions with converted exam scores.","GAOKAO-Bench 用中国高考主客观题评测 LLM，并把人类考试总分与主观题评分一致性纳入审计。"],"why":"It mixes objective and subjective Chinese exam scoring, making grader provenance and score conversion visible.","primary_link":"https://arxiv.org/abs/2305.12474","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenLMLab/GAOKAO-Bench"}],"link_count":3,"sections":9},{"id":"factscore-2023","title":"FActScore: Fine-grained Atomic Evaluation of Factual Precision in Long Form Text Generation","year":2023,"venue":"EMNLP 2023 main","authors":["Sewon Min","Kalpesh Krishna","Xinxi Lyu","Mike Lewis","Wen-tau Yih","Pang Wei Koh","Mohit Iyyer","Luke Zettlemoyer","Hannaneh Hajishirzi"],"authors_zh":"Sewon Min 等（University of Washington、University of Massachusetts Amherst、Meta AI、Allen Institute for AI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["factuality","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","factuality"],"status":"verified","priority":"可读","paper_type_zh":"EMNLP 2023 main 的 benchmark / evaluation surface","best_for_zh":"关注长文本事实性、grounding 评测、atomic claim 审计和自动 factuality metric 复用边界的研究者。","confidence":"high","one_line":["FActScore exposes atomic fact decomposition and retrieval-grounded factual precision as an auditable evaluation surface.","FActScore 将长回答拆成 atomic facts，并用知识源支持标签评估 factual precision。"],"why":"Good reference for grounding and factuality judge design.","primary_link":"https://aclanthology.org/2023.emnlp-main.741/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/shmsw25/FActScore"}],"link_count":6,"sections":9},{"id":"felm-factuality-evaluation-2023","title":"FELM: Benchmarking Factuality Evaluation of Large Language Models","year":2023,"venue":"NeurIPS 2023 Datasets and Benchmarks Track","authors":["Shiqi Chen","Yiran Zhao","Jinghan Zhang","I-Chun Chern","Siyang Gao","Pengfei Liu","Junxian He"],"authors_zh":"Shiqi Chen、Yiran Zhao、Jinghan Zhang、I-Chun Chern、Siyang Gao、Pengfei Liu、Junxian He","tracks":["judgment_rubric_domain_expert_data"],"source_role":["benchmark","data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["evaluation"],"construction_layer":["prompt_sourcing","reward_verifier_layer","release_audit"],"domains":["llm-evaluation"],"tags":["llm-judge","rubric","evaluation"],"status":"verified","priority":"可读","paper_type_zh":"评测、裁判或量规数据论文","best_for_zh":"需要构建、校准或审计开放式模型评测的研究者。","confidence":"high","one_line":["FELM benchmarks factuality evaluators with segment-level error labels, error types, and external evidence.","以带证据链接和错误类型的片段级标注，评测事实性裁判能否定位开放回答中的局部错误。"],"why":"FELM benchmarks factuality evaluators with segment-level error labels, error types, and external evidence.","primary_link":"https://papers.nips.cc/paper_files/paper/2023/hash/8b8a7960d343e023a6a0afe37eee6022-Abstract-Datasets_and_Benchmarks.html","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hkust-nlp/felm"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/hkust-nlp/felm"}],"link_count":3,"sections":9},{"id":"gaia-general-ai-assistants-2023","title":"GAIA: a benchmark for General AI Assistants","year":2023,"venue":"ICLR 2024 poster / arXiv","authors":["Grégoire Mialon","Clémentine Fourrier","Craig Swift","Thomas Wolf","Yann LeCun","Thomas Scialom"],"authors_zh":"Grégoire Mialon 等（FAIR, Meta、HuggingFace、AutoGPT、GenAI, Meta）","tracks":["benchmarks_evaluation_surfaces","environment_agent_trajectory_data"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["optimizer_scaffold","reward_verifier_layer"],"domains":["general-ai-assistants","interactive-agent-benchmark"],"tags":["benchmark","interactive_agent_benchmark","general-ai-assistants"],"status":"verified","priority":"必读","paper_type_zh":"ICLR 2024 poster / arXiv 的 general assistant benchmark","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["GAIA evaluates general assistants with 466 hidden-answer real-world questions requiring reasoning, tools, browsing, files, and multimodal handling.","GAIA 用 466 个现实 assistant 问题评测工具使用、多模态、网页浏览和推理的组合能力。"],"why":"It is a compact hidden-answer benchmark for general assistant orchestration across tools, web, files, and multimodal context.","primary_link":"https://arxiv.org/abs/2311.12983","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/gaia-benchmark/GAIA"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/gaia-benchmark"}],"link_count":4,"sections":9},{"id":"gpqa-google-proof-qa-2023","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","year":2023,"venue":"COLM 2024 / arXiv","authors":["David Rein","Betty Li Hou","Asa Cooper Stickland","Jackson Petty","Richard Yuanzhe Pang","Julien Dirani","Julian Michael","Samuel R. Bowman"],"authors_zh":"David Rein 等（New York University、Cohere、Anthropic PBC）","tracks":["benchmarks_evaluation_surfaces","judgment_rubric_domain_expert_data"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["science-reasoning","domain-expert-benchmark"],"tags":["benchmark","domain_expert_benchmark","science-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"COLM 2024 / arXiv 的 expert science QA benchmark","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["GPQA provides 448 expert-written Google-proof science multiple-choice questions for scalable oversight and hard-domain reasoning evaluation.","GPQA 提供 448 道专家编写的 Google-proof 科学选择题，用于困难领域推理和 scalable oversight 评测。"],"why":"It is a high-difficulty expert science benchmark for scalable oversight and contamination-sensitive evaluation.","primary_link":"https://arxiv.org/abs/2311.12022","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/idavidrein/gpqa"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/idavidrein/gpqa"}],"link_count":5,"sections":9},{"id":"ifeval-2023","title":"Instruction-Following Evaluation for Large Language Models","year":2023,"venue":"arXiv / Google Research","authors":["Jeffrey Zhou","Tianjian Lu","Swaroop Mishra","Siddhartha Brahma","Sujoy Basu","Yi Luan","Denny Zhou","Le Hou"],"authors_zh":"Jeffrey Zhou 等（Google Research）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["instruction-following","rule-based-evaluation"],"tags":["benchmark","instruction-following","rule-based-evaluation"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["IFEval 用规则可检查的指令约束评测模型是否真的遵循格式、长度、关键词、语言等要求。","IFEval 用规则可检查的指令约束评测模型是否真的遵循格式、长度、关键词、语言等要求。"],"why":"它提供 541 条 prompt，包含多类可程序检查的 instruction 约束。 的评测规模信息和 instruction-following 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2311.07911","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/google-research/google-research/tree/master/instruction_following_eval"}],"link_count":3,"sections":9},{"id":"evalplus-2023","title":"Is Your Code Generated by LLM Really Correct? Rigorous Evaluation of Large Language Models for Code Generation","year":2023,"venue":"NeurIPS 2023 / arXiv","authors":["Jiawei Liu","Chunqiu Steven Xia","Yuyao Wang","Lingming Zhang"],"authors_zh":"Jiawei Liu 等（University of Illinois Urbana-Champaign）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["code-generation","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","code-generation"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2023 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["EvalPlus exposes strengthened HumanEval+ and MBPP+ test suites as an auditable evaluation surface.","EvalPlus 将 strengthened HumanEval+ and MBPP+ test suites 做成可审计的评测面。"],"why":"Exposes false positives in saturated code benchmarks, strong audit angle.","primary_link":"https://arxiv.org/abs/2305.01210","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/evalplus/evalplus"},{"key":"project","label":["Project","项目主页"],"url":"https://evalplus.github.io/"}],"link_count":5,"sections":9},{"id":"mt-bench-chatbot-arena-2023","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","year":2023,"venue":"NeurIPS 2023 Datasets and Benchmarks / arXiv","authors":["Lianmin Zheng","Wei-Lin Chiang","Ying Sheng","Siyuan Zhuang","Zhanghao Wu","Yonghao Zhuang","Zi Lin","Zhuohan Li","Dacheng Li","Eric P. Xing","Hao Zhang","Joseph E. Gonzalez","Ion Stoica"],"authors_zh":"Lianmin Zheng 等（UC Berkeley、UC San Diego、Carnegie Mellon University、Stanford 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference","scalar_reward"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["chat-instruction","judge-reward-meta-evaluation"],"tags":["benchmark","judge_reward_meta_evaluation","chat-instruction"],"status":"verified","priority":"必读","paper_type_zh":"NeurIPS 2023 Datasets and Benchmarks / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MT-Bench / Chatbot Arena exposes GPT-4 pairwise/multi-turn judging and human arena votes as an auditable evaluation surface.","MT-Bench 与 Chatbot Arena 把 GPT-4 的成对及多轮评判和人类竞技场投票做成可审计的评测面。"],"why":"Canonical LLM-as-judge bias and agreement analysis.","primary_link":"https://arxiv.org/abs/2306.05685","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lm-sys/FastChat/tree/main/fastchat/llm_judge"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/lmsys/mt_bench_human_judgments"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/spaces/lmsys/chatbot-arena-leaderboard"}],"link_count":5,"sections":9},{"id":"nl2code-survey-2023","title":"Large Language Models Meet NL2Code: A Survey","year":2023,"venue":"ACL 2023 Long Papers","authors":["Daoguang Zan","Bei Chen","Fengji Zhang","Dianjie Lu","Bingchao Wu","Bei Guan","Wang Yongji","Jian-Guang Lou"],"authors_zh":"Daoguang Zan 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["programmatic"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["release_audit"],"domains":["nl2code","code-generation","evaluation"],"tags":["survey","nl2code","code-generation","evaluation","acl-2023"],"status":"verified","priority":"可读","paper_type_zh":"综述","best_for_zh":"希望选择代码生成基准或比较 NL2Code 模型的读者。","confidence":"high","one_line":["An ACL 2023 survey of LLMs for converting natural-language descriptions into code.","系统梳理大语言模型如何把自然语言需求转换为代码，并比较模型、基准与指标。"],"why":"It helps readers separate model scale, data quality, tuning, and executable evaluation when comparing NL2Code systems.","primary_link":"https://aclanthology.org/2023.acl-long.411/","links":[{"key":"project","label":["Project","项目主页"],"url":"https://nl2code.github.io"}],"link_count":3,"sections":9},{"id":"leandojo-benchmark-2023","title":"LeanDojo: Theorem Proving with Retrieval-Augmented Language Models","year":2023,"venue":"NeurIPS 2023 / arXiv","authors":["Kaiyu Yang","Aidan M. Swope","Alex Gu","Rahul Chalamala","Peiyang Song","Shixing Yu","Saad Godil","Ryan Prenger","Anima Anandkumar"],"authors_zh":"Kaiyu Yang 等（Caltech、NVIDIA、MIT、UC Santa Barbara 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["formal-proof-agents","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","formal-proof-agents"],"status":"verified","priority":"可读","paper_type_zh":"NeurIPS 2023 / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["LeanDojo Benchmark exposes Lean theorem proving with retrieval/tracing environment as an auditable evaluation surface.","LeanDojo Benchmark 把带检索与追踪环境的 Lean 定理证明做成可审计的评测面。"],"why":"Strong infrastructure lead for interactive Lean proof search.","primary_link":"https://arxiv.org/abs/2306.15626","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lean-dojo/LeanDojo"},{"key":"project","label":["Project","项目主页"],"url":"https://leandojo.org/"}],"link_count":4,"sections":9},{"id":"legalbench-2023","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","year":2023,"venue":"arXiv preprint / open-science project","authors":["Neel Guha","Julian Nyarko","Daniel E. Ho","Christopher Ré","Adam Chilton","Aditya Narayana","Alex Chohlas-Wood","Austin Peters","Brandon Waldon","Daniel N. Rockmore","Diego Zambrano","Dmitry Talisman","Enam Hoque","Faiz Surani","Frank Fagan","Galit Sarfaty","Gregory M. Dickinson","Haggai Porat","Jason Hegland","Jessica Wu","Joe Nudell","Joel Niklaus","John Nay","Jonathan H. Choi","Kevin Tobia","Margaret Hagan","Megan Ma","Michael Livermore","Nikon Rasumov-Rahe","Nils Holzenberger","Noam Kolt","Peter Henderson","Sean Rehaag","Sharad Goel","Shang Gao","Spencer Williams","Sunny Gandhi","Tom Zur","Varun Iyer","Zehua Li"],"authors_zh":"Neel Guha 等（Stanford University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["law","domain-expert-benchmark"],"tags":["benchmark","domain_expert_benchmark","law"],"status":"verified","priority":"可读","paper_type_zh":"arXiv preprint 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["LegalBench exposes legal reasoning task suite as an auditable evaluation surface.","LegalBench 将 legal reasoning task suite 做成可审计的评测面。"],"why":"Expert legal task taxonomy with clear domain specificity.","primary_link":"https://arxiv.org/abs/2308.11462","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/HazyResearch/legalbench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/nguha/legalbench"},{"key":"project","label":["Project","项目主页"],"url":"https://hazyresearch.stanford.edu/legalbench/"}],"link_count":5,"sections":9},{"id":"lets-verify-step-by-step-process-supervision-2023","title":"Let's Verify Step by Step","year":2023,"venue":"NeurIPS 2023","authors":["Hunter Lightman","Vineet Kosaraju","Yura Burda et al."],"authors_zh":"Lightman et al.","tracks":["process_trace_supervision_data"],"source_role":["verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["process_reward"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["process-trace-batch-2026","process-supervision"],"status":"verified","priority":"必读","paper_type_zh":"过程/轨迹监督数据与过程奖励研究","best_for_zh":"构建、审计或复用步骤级推理反馈数据的研究者。","confidence":"high","one_line":["Let's Verify Step by Step exposes process or trace supervision data.","发布 PRM800K：为 MATH 解题轨迹提供 80 万条人工步骤反馈标签，并实证过程监督优于只监督最终答案。"],"why":"It makes intermediate reasoning feedback auditable before reuse.","primary_link":"https://arxiv.org/abs/2305.20050","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/openai/prm800k"}],"link_count":2,"sections":9},{"id":"math-500-2023","title":"Let's Verify Step by Step (MATH-500 Evaluation Subset)","year":2023,"venue":"arXiv","authors":["Hunter Lightman","Vineet Kosaraju","Yura Burda","Harri Edwards","Bowen Baker","Teddy Lee","Jan Leike","John Schulman","Ilya Sutskever","Karl Cobbe"],"authors_zh":"Hunter Lightman 等（OpenAI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","process_supervision"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["math-evaluation","held-out-subset","process-supervision"],"tags":["benchmark","math-evaluation","held-out-subset","process-supervision"],"status":"partial","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"medium","one_line":["MATH-500 是 verifier、PRM 和推理模型报告里常用的固定评测子集，用来降低整套 MATH 评测成本。","MATH-500 是 verifier、PRM 和推理模型报告里常用的固定评测子集，用来降低整套 MATH 评测成本。"],"why":"它提供 500 道 MATH 题的常用 held-out 子集。 的评测规模信息和 math-evaluation 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2305.20050","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/prm800k"}],"link_count":3,"sections":9},{"id":"lima-2023","title":"LIMA: Less Is More for Alignment","year":2023,"venue":"NeurIPS 2023","authors":["Chunting Zhou","Pengfei Liu","Puxin Xu","Srini Iyer","Jiao Sun","Yuning Mao","Xuezhe Ma","Avia Efrat","Ping Yu","Lili Yu","Susan Zhang","Gargi Ghosh","Mike Lewis","Luke Zettlemoyer","Omer Levy"],"authors_zh":"Chunting Zhou, Pengfei Liu, Puxin Xu, Srini Iyer, Jiao Sun, Yuning Mao, Xuezhe Ma, Avia Efrat, Ping Yu, Lili Yu, Susan Zhang, Gargi Ghosh, Mike Lewis, Luke Zettlemoyer, Omer Levy","tracks":["foundations_and_primers"],"source_role":["data_release","construction_recipe","audit_failure"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","audit"],"construction_layer":["trace_writing","release_audit"],"domains":["large-language-models","practical-alignment-and-data-quality-primers"],"tags":["practical-alignment-and-data-quality-primers","arxiv-2305.11206","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"小数据监督对齐与数据质量研究","best_for_zh":"适合设计或审计小规模精选 instruction-tuning 数据集的读者。","confidence":"high","one_line":["LIMA uses 1,000 curated instruction-response pairs to test whether a capable base model needs data quality more than data volume to learn an assistant style.","LIMA 用 1,000 条精选指令—回答记录检验：能力较强的基础模型学习助手风格时，数据质量是否比数据规模更关键。"],"why":"It turns the quality-versus-quantity claim into a concrete small-data SFT comparison while keeping the preference-only evidence boundary visible.","primary_link":"https://arxiv.org/abs/2305.11206","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/GAIR/lima"}],"link_count":2,"sections":9},{"id":"longbench-2023","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","year":2023,"venue":"ACL 2024 long paper","authors":["Yushi Bai","Xin Lv","Jiajie Zhang","Hongchang Lyu","Jiankai Tang","Zhidian Huang","Zhengxiao Du","Xiao Liu","Aohan Zeng","Lei Hou","Yuxiao Dong","Jie Tang","Juanzi Li"],"authors_zh":"Yushi Bai 等（Tsinghua University、Zhipu.AI、Institute of Automation, Chinese Academy of Sciences）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["long-context","long-context-grounding"],"tags":["benchmark","long_context_grounding","long-context"],"status":"verified","priority":"可读","paper_type_zh":"ACL 2024 long paper 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["LongBench exposes bilingual long-document QA, summarization, few-shot, and code tasks as an auditable evaluation surface.","LongBench 把中英双语的长文档问答、摘要、少样本与代码任务做成可审计的评测面。"],"why":"Standard long-context suite; include as baseline.","primary_link":"https://aclanthology.org/2024.acl-long.172/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/THUDM/LongBench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/THUDM/LongBench"}],"link_count":5,"sections":9},{"id":"mathvista-2023","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","year":2023,"venue":"ICLR 2024 Oral / arXiv","authors":["Pan Lu","Hritik Bansal","Tony Xia","Jiacheng Liu","Chunyuan Li","Hannaneh Hajishirzi","Hao Cheng","Kai-Wei Chang","Michel Galley","Jianfeng Gao"],"authors_zh":"Pan Lu 等（University of California, Los Angeles、University of Washington、Microsoft Research）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["visual-math","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","visual-math"],"status":"verified","priority":"必读","paper_type_zh":"ICLR 2024 Oral / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MathVista exposes visual mathematical reasoning over diagrams, charts, and figures as an auditable evaluation surface.","MathVista 把基于示意图、图表与插图的视觉数学推理做成可审计的评测面。"],"why":"Canonical quantitative visual reasoning benchmark.","primary_link":"https://arxiv.org/abs/2310.02255","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/lupantech/MathVista"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/AI4Math/MathVista"},{"key":"project","label":["Project","项目主页"],"url":"https://mathvista.github.io/"}],"link_count":5,"sections":9},{"id":"mmmu-2023","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","year":2023,"venue":"CVPR 2024 Oral / arXiv","authors":["Xiang Yue","Yuansheng Ni","Kai Zhang","Tianyu Zheng","Ruoqi Liu","Ge Zhang","Samuel Stevens","Dongfu Jiang","Weiming Ren","Yuxuan Sun","Cong Wei","Botao Yu","Ruibin Yuan","Renliang Sun","Ming Yin","Boyuan Zheng","Zhenzhu Yang","Yibo Liu","Wenhao Huang","Huan Sun","Yu Su","Wenhu Chen"],"authors_zh":"Xiang Yue 等（IN.AI Research、University of Waterloo、The Ohio State University、Independent 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["multimodal-academic-reasoning","multimodal-reasoning-benchmark"],"tags":["benchmark","multimodal_reasoning_benchmark","multimodal-academic-reasoning"],"status":"verified","priority":"必读","paper_type_zh":"CVPR 2024 Oral / arXiv 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["MMMU exposes college-level multimodal subject questions as an auditable evaluation surface.","MMMU 把大学水平的多模态学科题目做成可审计的评测面。"],"why":"Broad domain reasoning with heterogeneous images.","primary_link":"https://arxiv.org/abs/2311.16502","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/MMMU-Benchmark/MMMU"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/MMMU/MMMU"},{"key":"project","label":["Project","项目主页"],"url":"https://mmmu-benchmark.github.io/"}],"link_count":6,"sections":9},{"id":"openassistant-conversations-democratizing-large-language-model-alignment-2023","title":"OpenAssistant Conversations - Democratizing Large Language Model Alignment","year":2023,"venue":"arXiv","authors":["Andreas Köpf","Yannic Kilcher","Dimitri von Rütte","Sotiris Anagnostidis","Zhi-Rui Tam","Keith Stevens","Abdullah Barhoum","Nguyen Minh Duc","Oliver Stanley","Richárd Nagyfi","Shahul ES","Sameer Suri","David Glushkov","Arnav Dantuluri","Andrew Maguire","Christoph Schuhmann","Huu Nguyen","Alexander Mattick"],"authors_zh":"Andreas Köpf、Yannic Kilcher、Dimitri von Rütte、Sotiris Anagnostidis、Zhi-Rui Tam、Keith Stevens、Abdullah Barhoum、Nguyen Minh Duc、Oliver Stanley、Richárd Nagyfi、Shahul ES、Sameer Suri、David Glushkov、Arnav Dantuluri、Andrew Maguire、Christoph Schuhmann、Huu Nguyen、Alexander Mattick","tracks":["preference_reward_feedback_data"],"source_role":["data_release"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["llm-post-training"],"tags":["candidate-batch","post-training","reward-or-judgment"],"status":"verified","priority":"可读","paper_type_zh":"后训练数据、偏好、奖励或评测研究","best_for_zh":"研究 LLM 后训练反馈数据与 verifier 的读者。","confidence":"high","one_line":["OpenAssistant Conversations - Democratizing Large Language Model Alignment addresses open conversational alignment data.","OpenAssistant Conversations 构建并公开了 OASST1，包含约 16.1 万条消息、35 种语言、46.1 万个质量评分和一万余棵完整对话树，可用于指令微调、奖励模型训练和人类偏好对齐。"],"why":"It exposes a feedback, preference, reward, rubric, safety, or post-training data surface that must be audited before reuse.","primary_link":"https://arxiv.org/abs/2304.07327","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/LAION-AI/Open-Assistant"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/OpenAssistant/oasst1"},{"key":"project","label":["Project","项目主页"],"url":"https://open-assistant.io/"}],"link_count":4,"sections":9},{"id":"orca-2-2023","title":"Orca 2: Teaching Small Language Models How to Reason","year":2023,"venue":"arXiv preprint (2023)","authors":["Arindam Mitra","Luciano Del Corro","Shweti Mahajan","Andres Codas","Clarisse Simoes","Sahaj Agarwal","Xuxi Chen","Anastasia Razdaibiedina","Erik Jones","Kriti Aggarwal","Hamid Palangi","Guoqing Zheng","Corby Rosset","Hamed Khanpour","Ahmed Awadallah"],"authors_zh":"Arindam Mitra, Luciano Del Corro, Shweti Mahajan, Andres Codas, Clarisse Simoes, Sahaj Agarwal, Xuxi Chen, Anastasia Razdaibiedina, Erik Jones, Kriti Aggarwal, Hamid Palangi, Guoqing Zheng, Corby Rosset, Hamed Khanpour, Ahmed Awadallah","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["trace_writing"],"domains":["large-language-models","teacher-explanation-distillation"],"tags":["teacher-explanation-distillation","prompt-erasing","arxiv-2311.11045","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"教师推理策略蒸馏","best_for_zh":"适合为小模型设计 teacher 生成推理示范的读者。","confidence":"high","one_line":["Orca 2 elicits task-specific reasoning strategies from GPT-4 and erases the strategy prompt during SFT so a smaller model learns both how and when to reason.","Orca 2 让 GPT-4 按任务生成不同策略的推理回答，并在 SFT 时擦除策略提示，使小模型同时学习怎样推理和何时采用哪种策略。"],"why":"It changes teacher imitation from copying one response style to training on strategy-tailored answer traces.","primary_link":"https://arxiv.org/abs/2311.11045","links":[{"key":"project","label":["Project","项目主页"],"url":"https://aka.ms/orca-lm"}],"link_count":2,"sections":9},{"id":"raft-reward-ranked-finetuning-2023","title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","year":2023,"venue":"arXiv preprint","authors":["Junxian He","Chunting Zhou","Xuezhe Ma","Taylor Berg-Kirkpatrick","Graham Neubig"],"authors_zh":"Junxian He、Chunting Zhou、Xuezhe Ma、Taylor Berg-Kirkpatrick、Graham Neubig","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["scalar_reward"],"training_use":["sft","reward_modeling"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","text_generation"],"tags":["raft","reward-ranking","self-training","alignment"],"status":"verified","priority":"必读","paper_type_zh":"奖励排序驱动的数据选择与生成模型对齐研究","best_for_zh":"适合研究奖励模型如何作为训练样本选择器的读者。","confidence":"high","one_line":["RAFT samples candidates, selects the reward-best response, and fine-tunes the policy on it.","RAFT 从多个策略候选中保留奖励最高的回答，并用该回答迭代监督微调模型。"],"why":"It turns a reward model into a data-selection mechanism rather than an online RL objective.","primary_link":"https://arxiv.org/abs/2304.06767","links":[],"link_count":2,"sections":9},{"id":"reasoning-prompting-survey-2023","title":"Reasoning with Language Model Prompting: A Survey","year":2023,"venue":"ACL 2023 (Long Papers)","authors":["Shuofei Qiao","Yixin Ou","Ningyu Zhang","Xiang Chen","Yunzhi Yao","Shumin Deng","Chuanqi Tan","Fei Huang","Huajun Chen"],"authors_zh":"Shuofei Qiao 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["language-model-prompting","reasoning","benchmarks"],"tags":["foundations-and-primers","reasoning","prompting","acl-2023","survey"],"status":"verified","priority":"可读","paper_type_zh":"提示推理综述","best_for_zh":"想系统了解提示式推理方法、任务与比较方式的读者。","confidence":"high","one_line":["An ACL 2023 survey of prompting methods used to elicit and study reasoning in language models.","系统梳理如何用提示激发语言模型推理，并提供方法比较与学习资源。"],"why":"It helps readers compare reasoning methods by what the prompt changes, which task is measured, and what evidence supports the claim.","primary_link":"https://aclanthology.org/2023.acl-long.294/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/zjunlp/Prompt4ReasoningPapers"}],"link_count":3,"sections":9},{"id":"rrhf-rank-responses-2023","title":"RRHF: Rank Responses to Align Language Models with Human Feedback","year":2023,"venue":"NeurIPS 2023","authors":["Zheng Yuan","Hongyi Yuan","Chuanqi Tan","Wei Wang","Songfang Huang","Fei Huang"],"authors_zh":"Zheng Yuan、Hongyi Yuan、Chuanqi Tan、Wei Wang、Songfang Huang、Fei Huang（阿里巴巴集团）","tracks":["training_usage_optimization_objectives"],"source_role":["construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["sft","preference_learning"],"construction_layer":["optimizer_scaffold"],"domains":["alignment","dialogue"],"tags":["rrhf","response-ranking","preference-learning","alignment"],"status":"verified","priority":"可读","paper_type_zh":"排序响应驱动的轻量级偏好对齐研究","best_for_zh":"适合理解多候选回答与排序反馈如何直接进入训练目标的读者。","confidence":"high","one_line":["RRHF trains a policy to rank sampled responses in the same order as feedback, using a lightweight ranking loss.","RRHF 让模型对多来源候选回答的概率排序与反馈排序一致，无需 PPO 或单独奖励模型。"],"why":"It demonstrates that the value of feedback can be carried by an ordered candidate set rather than only by a reward model or a binary pair.","primary_link":"https://arxiv.org/abs/2304.05302","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/GanjinZero/RRHF"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Anthropic/hh-rlhf"}],"link_count":4,"sections":9},{"id":"self-instruct-2023","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","year":2023,"venue":"ACL 2023","authors":["Yizhong Wang","Yeganeh Kordi","Swaroop Mishra","Alisa Liu","Noah A. Smith","Daniel Khashabi","Hannaneh Hajishirzi"],"authors_zh":"Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, Alisa Liu, Noah A. Smith, Daniel Khashabi, Hannaneh Hajishirzi","tracks":["instruction_demonstration_rationale_data"],"source_role":["data_release","construction_recipe"],"verification_contract":["judgment_required"],"supervision_granularity":["answer_level"],"training_use":["sft","distillation"],"construction_layer":["prompt_sourcing","trace_writing","release_audit"],"domains":["large-language-models","synthetic-instruction-generation"],"tags":["synthetic-instruction-generation","arxiv-2212.10560","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"合成指令数据生成与发布","best_for_zh":"适合构建自生成 instruction SFT 数据，或审计任务多样性与 teacher 错误的读者。","confidence":"high","one_line":["Self-Instruct repeatedly asks GPT-3 to invent, filter, and answer new tasks, expanding 175 human seeds into 82,439 synthetic SFT instances.","Self-Instruct 反复让 GPT-3 发明、过滤并回答新任务，把 175 条人写种子扩展为 82,439 条合成 SFT 实例。"],"why":"It provides the canonical inspectable loop for bootstrapping instruction data while exposing the quality limits of self-generated targets.","primary_link":"https://arxiv.org/abs/2212.10560","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/yizhongw/self-instruct"},{"key":"data","label":["Data","数据"],"url":"https://github.com/yizhongw/self-instruct/tree/main/data"}],"link_count":3,"sections":9},{"id":"swe-bench-github-issues-2023","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","year":2023,"venue":"ICLR 2024 Oral / arXiv","authors":["Carlos E. Jimenez","John Yang","Alexander Wettig","Shunyu Yao","Kexin Pei","Ofir Press","Karthik Narasimhan"],"authors_zh":"Carlos E. Jimenez 等（Princeton University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["swe-agents","repository-repair","github-issues"],"tags":["agent_environment","trajectory_data","swe-agents","repository-repair","github-issues"],"status":"verified","priority":"必读","paper_type_zh":"ICLR 2024 Oral / arXiv 的 SWE agent benchmark","best_for_zh":"关注终端、SWE、桌面、办公自动化和专业工作智能体环境与轨迹数据的研究者。","confidence":"high","one_line":["SWE-bench turns real GitHub issues into executable repository repair tasks.","SWE-bench 将真实 GitHub issue 转化为可执行仓库修复任务。"],"why":"it is the foundational executable repository repair benchmark","primary_link":"https://arxiv.org/abs/2310.06770","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/SWE-bench/SWE-bench"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/princeton-nlp/SWE-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://www.swebench.com/"}],"link_count":6,"sections":9},{"id":"toollm-toolbench-2023","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","year":2023,"venue":"ICLR 2024 spotlight","authors":["Yujia Qin","Shihao Liang","Yining Ye","Kunlun Zhu","Lan Yan","Yaxi Lu","Yankai Lin","Xin Cong","Xiangru Tang","Bill Qian","Sihan Zhao","Lauren Hong","Runchu Tian","Ruobing Xie","Jie Zhou","Mark Gerstein","Dahai Li","Zhiyuan Liu","Maosong Sun"],"authors_zh":"Yujia Qin 等（Tsinghua University、ModelBest Inc.、Renmin University of China、Yale University、WeChat AI, Tencent Inc.、Zhihu Inc.）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["data_release","benchmark","agent_environment"],"verification_contract":["environmental","programmatic"],"supervision_granularity":["state_action_level","full_episode"],"training_use":["sft","agent_training","evaluation","audit"],"construction_layer":["search_substrate","reward_verifier_layer","release_audit"],"domains":["tool-use","agent-environments","interactive-evaluation"],"tags":["agent-environment","benchmark","evaluation-surface","tool-use","trajectory-data"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2024 spotlight 的评测面 / 环境基准","best_for_zh":"关注工具使用数据构造、多工具调用轨迹、RapidAPI 漂移和自动评审耦合风险的读者。","confidence":"medium","one_line":["ToolLLM introduces ToolBench, a large tool-use dataset over 16,464 real-world APIs, plus ToolEval and ToolLLaMA for training and evaluating API-using models.","ToolLLM 提出 ToolBench：覆盖 16,464 个真实 API 的工具使用数据集，并配套 ToolEval 与 ToolLLaMA 用于训练和评测。"],"why":"ToolLLM introduces ToolBench, a large tool-use dataset over 16,464 real-world APIs, plus ToolEval and ToolLLaMA for training and evaluating API-using models.","primary_link":"https://arxiv.org/abs/2307.16789","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/OpenBMB/ToolBench"},{"key":"project","label":["Project","项目主页"],"url":"https://openbmb.github.io/ToolBench/"}],"link_count":5,"sections":9},{"id":"reasoning-in-llms-survey-2023","title":"Towards Reasoning in Large Language Models: A Survey","year":2023,"venue":"Findings of ACL 2023","authors":["Jie Huang","Kevin Chen-Chuan Chang"],"authors_zh":"Jie Huang, Kevin Chen-Chuan Chang","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["mixed"],"supervision_granularity":["unknown"],"training_use":["evaluation","audit"],"construction_layer":["release_audit"],"domains":["large-language-models","field-surveys"],"tags":["field-surveys","arxiv-2212.10403","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"LLM 推理领域综述与决策地图","best_for_zh":"适合筛选推理论文，或区分数据、提示、训练、评测与解释主张的研究者。","confidence":"high","one_line":["This survey separates supervised reasoning, inference-time elicitation, self-improvement, evaluation, and interpretation into distinct claim types.","这篇综述把监督推理、推理时诱导、自改进、评测和解释拆成不同主张类型。"],"why":"It prevents prompt gains, training effects, benchmark scores, and faithful-reasoning claims from being conflated.","primary_link":"https://arxiv.org/abs/2212.10403","links":[{"key":"project","label":["Project","项目主页"],"url":"https://github.com/jeffhj/LM-reasoning"}],"link_count":3,"sections":9},{"id":"webarena-realistic-web-environment-2023","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","year":2023,"venue":"Published as ICLR 2024 in arXiv PDF; WebArena-x index also labels WebArena as NeurIPS 2024 Oral","authors":["Shuyan Zhou","Frank F. Xu","Hao Zhu","Xuhui Zhou","Robert Lo","Abishek Sridhar","Xianyi Cheng","Tianyue Ou","Yonatan Bisk","Daniel Fried","Uri Alon","Graham Neubig"],"authors_zh":"Shuyan Zhou 等（Carnegie Mellon University、Inspired Cognition）","tracks":["environment_agent_trajectory_data","benchmarks_evaluation_surfaces"],"source_role":["benchmark","agent_environment"],"verification_contract":["environmental"],"supervision_granularity":["full_episode","state_action_level"],"training_use":["evaluation","audit"],"construction_layer":["optimizer_scaffold","reward_verifier_layer"],"domains":["web-agents","interactive-agent-benchmark"],"tags":["benchmark","interactive_agent_benchmark","web-agents"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2024","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"high","one_line":["WebArena exposes browser tasks on self-hosted realistic websites with functional checks as an auditable evaluation surface.","WebArena 把自建真实网站上带功能性检查的浏览器任务做成可审计的评测面。"],"why":"Canonical reproducible web-agent benchmark.","primary_link":"https://arxiv.org/abs/2307.13854","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/web-arena-x/webarena"},{"key":"project","label":["Project","项目主页"],"url":"https://webarena.dev/og/"}],"link_count":4,"sections":9},{"id":"big-bench-2022","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","year":2022,"venue":"Transactions on Machine Learning Research / arXiv","authors":["Aarohi Srivastava","Abhinav Rastogi","Abhishek Rao","BIG-bench authors"],"authors_zh":"Aarohi Srivastava 等（BIG-bench consortium: 132 institutions verified by official paper 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["general-reasoning","static-reasoning-benchmark"],"tags":["benchmark","static_reasoning_benchmark","general-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"TMLR 与 arXiv 的评测基准与评测面","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["BIG-bench exposes collaborative suite of diverse language-model tasks as an auditable evaluation surface.","BIG-bench 把社区协作构建的多样化语言模型任务集做成可审计的评测面。"],"why":"Canonical broad benchmark; lower priority due age and heterogeneity.","primary_link":"https://arxiv.org/abs/2206.04615","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/google/BIG-bench"},{"key":"project","label":["Project","项目主页"],"url":"https://github.com/google/BIG-bench/blob/main/docs/doc.md"}],"link_count":5,"sections":9},{"id":"ds-1000-2022","title":"DS-1000: A Natural and Reliable Benchmark for Data Science Code Generation","year":2022,"venue":"ICML 2023","authors":["Yuhang Lai","Chengxi Li","Yiming Wang","Tianyi Zhang","Ruiqi Zhong","Luke Zettlemoyer","Scott Wen-tau Yih","Daniel Fried","Sida Wang","Tao Yu"],"authors_zh":"Yuhang Lai 等（The University of Hong Kong、Peking University、Stanford University、UC Berkeley 等）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["reward_verifier_layer","release_audit"],"domains":["data-science-code","code-executable-benchmark"],"tags":["benchmark","code_executable_benchmark","data-science-code"],"status":"verified","priority":"可读","paper_type_zh":"ICML 2023 的 benchmark / evaluation surface","best_for_zh":"关注 benchmark、评分契约、污染风险、评测 harness 和后训练反馈可复用性的研究者。","confidence":"medium","one_line":["DS-1000 exposes data-science code completion with executable tests and API constraints as an auditable evaluation surface.","DS-1000 把带可执行测试与 API 约束的数据科学代码补全做成可审计的评测面。"],"why":"Real library-use surface across common Python data-science stacks.","primary_link":"https://arxiv.org/abs/2211.11501","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/xlang-ai/DS-1000"},{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/xlangai/DS-1000"},{"key":"project","label":["Project","项目主页"],"url":"https://ds1000-code-gen.github.io/"}],"link_count":5,"sections":9},{"id":"helm-2022","title":"Holistic Evaluation of Language Models","year":2022,"venue":"Transactions on Machine Learning Research / arXiv","authors":["Percy Liang","Rishi Bommasani","Tony Lee","Dimitris Tsipras","Dilara Soylu","Michihiro Yasunaga","Yian Zhang","Deepak Narayanan","Yuhuai Wu","Ananya Kumar"],"authors_zh":"Percy Liang 等（Stanford CRFM, Stanford University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","infrastructure"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["evaluation-framework","holistic-evaluation","benchmark-infrastructure"],"tags":["benchmark","evaluation-framework","holistic-evaluation","benchmark-infrastructure"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["HELM 把语言模型评测组织成“场景 × 指标”的矩阵，不只看准确率，也看校准、鲁棒性、公平性、毒性、效率等维度。","HELM 把语言模型评测组织成“场景 × 指标”的矩阵，不只看准确率，也看校准、鲁棒性、公平性、毒性、效率等维度。"],"why":"它提供 HELM 是 scenario × metric 的评测框架；初版报告覆盖数十个 scenarios、多种 metrics 和多类模型，具体规模随 HELM release 变化。 的评测规模信息和 evaluation-framework 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2211.09110","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/stanford-crfm/helm"},{"key":"project","label":["Project","项目主页"],"url":"https://crfm.stanford.edu/helm/latest/"}],"link_count":5,"sections":9},{"id":"zero-shot-cot-2022","title":"Large Language Models are Zero-Shot Reasoners","year":2022,"venue":"NeurIPS 2022","authors":["Takeshi Kojima","Shixiang Shane Gu","Machel Reid","Yutaka Matsuo","Yusuke Iwasawa"],"authors_zh":"Takeshi Kojima, Shixiang Shane Gu, Machel Reid, Yutaka Matsuo, Yusuke Iwasawa","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe"],"verification_contract":["programmatic","judgment_required"],"supervision_granularity":["step_level","answer_level"],"training_use":["sft","evaluation"],"construction_layer":["trace_writing"],"domains":["large-language-models","chain-of-thought-and-rationale-targets"],"tags":["chain-of-thought-and-rationale-targets","arxiv-2205.11916","primary-link-checked"],"status":"verified","priority":"可读","paper_type_zh":"思维链提示 / 理由数据论文","best_for_zh":"适合希望系统理解思维链与理由目标、数据对象与反馈契约的读者。","confidence":"high","one_line":["Zero-shot-CoT uses a generic cue and a second answer-extraction call to elicit test-time rationales without demonstrations.","Zero-shot-CoT 用通用提示和第二次答案抽取，在没有示例时诱导可见的测试时理由。"],"why":"It defines a minimal visible-rationale baseline while exposing the gap between answer accuracy and step validity.","primary_link":"https://arxiv.org/abs/2205.11916","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/kojima-takeshi188/zero_shot_cot"}],"link_count":2,"sections":9},{"id":"pretrain-prompt-predict-survey-2022","title":"Pre-train, Prompt, and Predict: A Systematic Survey of Prompting Methods in Natural Language Processing","year":2022,"venue":"ACM Computing Surveys","authors":["Pengfei Liu","Weizhe Yuan","Jinlan Fu","Zhengbao Jiang","Hiroaki Hayashi","Graham Neubig"],"authors_zh":"Pengfei Liu 等","tracks":["foundations_and_primers"],"source_role":["survey_background"],"verification_contract":["unknown"],"supervision_granularity":["unknown"],"training_use":["evaluation"],"construction_layer":["trace_writing"],"domains":["prompting","reasoning","reasoning-data","few-shot-learning","evaluation"],"tags":["foundations-and-primers","prompting","reasoning","high-citation","survey"],"status":"verified","priority":"必读","paper_type_zh":"提示学习系统综述","best_for_zh":"设计提示衍生推理数据或比较提示研究的读者。","confidence":"high","one_line":["A high-impact systematic survey of the prompt designs that later reasoning prompting builds on.","系统整理提示构造、预训练模型与微调策略的高引综述。"],"why":"It makes a prompt a documented transformation from task data to an answer space, not a casual string.","primary_link":"https://dl.acm.org/doi/10.1145/3560815","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/thunlp/PromptPapers"},{"key":"project","label":["Project","项目主页"],"url":"http://pretrain.nlpedia.ai/"}],"link_count":5,"sections":9},{"id":"star-2022","title":"STaR: Bootstrapping Reasoning With Reasoning","year":2022,"venue":"NeurIPS 2022","authors":["Eric Zelikman","Yuhuai Wu","Jesse Mu","Noah D. Goodman"],"authors_zh":"Eric Zelikman, Yuhuai Wu, Jesse Mu, Noah D. Goodman","tracks":["instruction_demonstration_rationale_data"],"source_role":["construction_recipe","scaling_study"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level","step_level"],"training_use":["sft","distillation"],"construction_layer":["self_play_anchor","trace_writing"],"domains":["large-language-models","self-training-demonstrations"],"tags":["self-training-demonstrations","arxiv-2203.14465","primary-link-checked"],"status":"verified","priority":"必读","paper_type_zh":"迭代式 rationale 自训练","best_for_zh":"适合用有标准答案但缺少书面 rationale 的题目构造推理 SFT 数据的研究者。","confidence":"high","one_line":["STaR repeatedly turns answer-correct model rationales into new SFT data, bootstrapping reasoning from few demonstrations.","STaR 反复把答案正确的模型 rationale 转成新一轮 SFT 数据，用少量示范自举推理能力。"],"why":"It shows how answer keys can replace large-scale rationale annotation when terminal correctness is reliable.","primary_link":"https://arxiv.org/abs/2203.14465","links":[],"link_count":1,"sections":9},{"id":"hh-rlhf-helpful-harmless-2022","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","year":2022,"venue":"arXiv preprint","authors":["Yuntao Bai","Andy Jones","Kamalloo Eshghian","Anna Askell","Yudong Chen","Nova DasSarma","Dawn Drain","Stanislav Fort","Deep Ganguli","Tom Henighan","Nicholas Joseph","Sushma Kadavath","Jackson Kernion","Tom Conerly","Sheer El-Showk","Nelson Elhage","Zac Hatfield-Dodds","Danny Hernandez","Tristan Hume","Scott Johnston","Shauna Kravec","Liane Lovitt","Neel Nanda","Catherine Olsson","Dario Amodei","Tom Brown","Jack Clark","Sam McCandlish","Chris Olah","Ben Mann","Jared Kaplan"],"authors_zh":"Yuntao Bai、Andy Jones、Kamalloo Eshghian、Anna Askell、Yudong Chen、Nova DasSarma、Dawn Drain、Stanislav Fort、Deep Ganguli、Tom Henighan、Nicholas Joseph、Sushma Kadavath、Jackson Kernion、Tom Conerly、Sheer El-Showk、Nelson Elhage、Zac Hatfield-Dodds、Danny Hernandez、Tristan Hume、Scott Johnston、Shauna Kravec、Liane Lovitt、Neel Nanda、Catherine Olsson、Dario Amodei、Tom Brown、Jack Clark、Sam McCandlish、Chris Olah、Ben Mann、Jared Kaplan","tracks":["preference_reward_feedback_data"],"source_role":["data_release","verifier_reward"],"verification_contract":["judgment_required"],"supervision_granularity":["pairwise_preference"],"training_use":["preference_learning","reward_modeling","evaluation","audit"],"construction_layer":["reward_verifier_layer"],"domains":["preference_reward_feedback_data"],"tags":["preference","reward-modeling","feedback-data"],"status":"verified","priority":"必读","paper_type_zh":"数据集论文","best_for_zh":"偏好学习、奖励建模与反馈数据审计","confidence":"","one_line":["HH-RLHF releases dialogue contexts, paired assistant responses, and helpfulness or harmlessness preferences including red-team interactions for bounded preference learning, reward modeling, evaluation, and audit.","HH-RLHF 发布围绕其任务场景组织的候选回答比较与反馈记录，可用于有边界的偏好学习、奖励建模、评测和审计。"],"why":"The release supports feedback learning for 对话回答的有帮助性和无害性.","primary_link":"https://arxiv.org/abs/2204.05862","links":[{"key":"data","label":["Data","数据"],"url":"https://huggingface.co/datasets/Anthropic/hh-rlhf"}],"link_count":2,"sections":9},{"id":"humaneval-2021","title":"Evaluating Large Language Models Trained on Code","year":2021,"venue":"arXiv / OpenAI","authors":["Mark Chen","Jerry Tworek","Heewoo Jun","Qiming Yuan","Henrique Ponde de Oliveira Pinto","Jared Kaplan","Harri Edwards","Yuri Burda","Nicholas Joseph","Greg Brockman","Alex Ray","Raul Puri","Gretchen Krueger","Michael Petrov","Heidy Khlaaf","Girish Sastry","Pamela Mishkin","Brooke Chan","Scott Gray","Nick Ryder","Mikhail Pavlov","Alethea Power","Lukasz Kaiser","Mohammad Bavarian","Clemens Winter","Philippe Tillet","Felipe Petroski Such","Dave Cummings","Matthias Plappert","Fotios Chantzis","Elizabeth Barnes","Ariel Herbert-Voss","William Hebgen Guss","Alex Nichol","Alex Paino","Nikolas Tezak","Jie Tang","Igor Babuschkin","Suchir Balaji","Shantanu Jain","William Saunders","Christopher Hesse","Andrew N. Carr","Jan Leike","Josh Achiam","Vedant Misra","Evan Morikawa","Alec Radford","Matthew Knight","Miles Brundage","Mira Murati","Katie Mayer","Peter Welinder","Bob McGrew","Dario Amodei","Sam McCandlish","Ilya Sutskever","Wojciech Zaremba"],"authors_zh":"Mark Chen 等（OpenAI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["code-generation","unit-test-benchmark","program-synthesis"],"tags":["benchmark","code-generation","unit-test-benchmark","program-synthesis"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["HumanEval 用小而精的函数合成任务和单元测试定义代码生成能力，是代码 benchmark 的基础坐标。","HumanEval 用小而精的函数合成任务和单元测试定义代码生成能力，是代码 benchmark 的基础坐标。"],"why":"它提供 164 道手写 Python 函数补全题。 的评测规模信息和 code-generation 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2107.03374","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/human-eval"}],"link_count":3,"sections":9},{"id":"apps-2021","title":"Measuring Coding Challenge Competence With APPS","year":2021,"venue":"NeurIPS 2021 Datasets and Benchmarks / arXiv","authors":["Dan Hendrycks","Steven Basart","Saurav Kadavath","Mantas Mazeika","Akul Arora","Ethan Guo","Collin Burns","Samir Puranik","Horace He","Dawn Song","Jacob Steinhardt"],"authors_zh":"Dan Hendrycks 等（UC Berkeley, University of Chicago）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["competitive-programming","code-executable-benchmark"],"tags":["benchmark","competitive-programming","code-executable-benchmark"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["APPS 把代码评测推进到竞赛编程和复杂输入输出题，能暴露长程算法推理与实现能力。","APPS 把代码评测推进到竞赛编程和复杂输入输出题，能暴露长程算法推理与实现能力。"],"why":"它提供 约 10,000 道编程竞赛题，按 introductory/interview/competition 难度组织。 的评测规模信息和 competitive-programming 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2105.09938","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hendrycks/apps"},{"key":"huggingface","label":["Hugging Face","Hugging Face"],"url":"https://huggingface.co/datasets/codeparrot/apps"}],"link_count":4,"sections":9},{"id":"math-dataset-2021","title":"Measuring Mathematical Problem Solving With the MATH Dataset","year":2021,"venue":"NeurIPS 2021 Datasets and Benchmarks / arXiv","authors":["Dan Hendrycks","Collin Burns","Saurav Kadavath","Akul Arora","Steven Basart","Eric Tang","Dawn Song","Jacob Steinhardt"],"authors_zh":"Dan Hendrycks 等（UC Berkeley, University of Chicago）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["competition-math","math-reasoning","answer-verification"],"tags":["benchmark","competition-math","math-reasoning","answer-verification"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["MATH 将竞赛级数学题系统化，覆盖代数、几何、数论、组合等方向，是硬数学评测的根节点之一。","MATH 将竞赛级数学题系统化，覆盖代数、几何、数论、组合等方向，是硬数学评测的根节点之一。"],"why":"它提供 12,500 道竞赛数学题，常用划分为 7,500 train / 5,000 test。 的评测规模信息和 competition-math 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2103.03874","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hendrycks/math"}],"link_count":3,"sections":9},{"id":"gsm8k-2021","title":"Training Verifiers to Solve Math Word Problems","year":2021,"venue":"arXiv / OpenAI","authors":["Karl Cobbe","Vineet Kosaraju","Mohammad Bavarian","Mark Chen","Heewoo Jun","Lukasz Kaiser","Matthias Plappert","Jerry Tworek","Jacob Hilton","Reiichiro Nakano","Christopher Hesse","John Schulman"],"authors_zh":"Karl Cobbe 等（OpenAI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark","verifier_reward"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["grade-school-math","math-word-problems","verifier-evaluation"],"tags":["benchmark","grade-school-math","math-word-problems","verifier-evaluation"],"status":"verified","priority":"必读","paper_type_zh":"OpenAI / arXiv 的 math word-problem benchmark","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["GSM8K provides 8,792 grade-school math word problems and uses final numeric answer checking while studying verifier models.","GSM8K 提供 8,792 道小学数学应用题，用最终数值答案检查，并研究 verifier models。"],"why":"It is a canonical answer-verifiable math benchmark and verifier-selection reference point.","primary_link":"https://arxiv.org/abs/2110.14168","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/openai/grade-school-math"}],"link_count":3,"sections":9},{"id":"truthfulqa-2021","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","year":2021,"venue":"ACL 2022 / arXiv","authors":["Stephanie Lin","Jacob Hilton","Owain Evans"],"authors_zh":"Stephanie Lin 等（University of Oxford, OpenAI）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["truthfulness","factuality","safety-evaluation"],"tags":["benchmark","truthfulness","factuality","safety-evaluation"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["TruthfulQA 检查模型是否复述常见误解和虚假信念，而不是给出真实回答。","TruthfulQA 检查模型是否复述常见误解和虚假信念，而不是给出真实回答。"],"why":"它提供 817 个问题，覆盖 38 个类别。 的评测规模信息和 truthfulness 方向的可复用 benchmark 坐标。","primary_link":"https://aclanthology.org/2022.acl-long.229/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sylinrl/TruthfulQA"}],"link_count":4,"sections":9},{"id":"mmlu-2020","title":"Measuring Massive Multitask Language Understanding","year":2020,"venue":"ICLR 2021 / arXiv","authors":["Dan Hendrycks","Collin Burns","Steven Basart","Andy Zou","Mantas Mazeika","Dawn Song","Jacob Steinhardt"],"authors_zh":"Dan Hendrycks 等（UC Berkeley, University of Chicago）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["broad-academic","static-reasoning-benchmark"],"tags":["benchmark","broad-academic","static-reasoning-benchmark"],"status":"verified","priority":"必读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要了解 benchmark 规模、scorer 契约、split/contamination 风险，并把评测结果用于 reasoning-data 审计的读者。","confidence":"high","one_line":["MMLU 建立了覆盖 57 个学科的大规模多选题评测面，用来衡量模型在预训练知识、专业考试和跨学科推理上的基本能力。","MMLU 建立了覆盖 57 个学科的大规模多选题评测面，用来衡量模型在预训练知识、专业考试和跨学科推理上的基本能力。"],"why":"它提供 约 15,908 道题，覆盖 57 个学科/任务。 的评测规模信息和 broad-academic 方向的可复用 benchmark 坐标。","primary_link":"https://arxiv.org/abs/2009.03300","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/hendrycks/test"}],"link_count":3,"sections":9},{"id":"medqa-usmle-2020","title":"What Disease Does This Patient Have? A Large-scale Open Domain Question Answering Dataset from Medical Exams","year":2020,"venue":"arXiv","authors":["Di Jin","Eileen Pan","Nassim Oufattole","Wei-Hung Weng","Hanyi Fang","Peter Szolovits"],"authors_zh":"Di Jin 等（MIT, Harvard Medical School）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["medical-exams","domain-expert-benchmark","multiple-choice"],"tags":["benchmark","medical-exams","domain-expert-benchmark","multiple-choice"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要补齐基础 benchmark 坐标、评测契约、contamination 和 scorer 审计的读者。","confidence":"high","one_line":["MedQA-USMLE provides 12,723 English medical licensing-exam questions, with the full MedQA release spanning 61,097 multilingual questions.","MedQA-USMLE 常用英文子集有 12,723 道医学执照考试选择题；完整 MedQA 多语 release 约 61,097 题。"],"why":"It supplies a reusable evaluation coordinate for medical-exams, domain-expert-benchmark work.","primary_link":"https://arxiv.org/abs/2009.13081","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/jind11/MedQA"}],"link_count":3,"sections":9},{"id":"hellaswag-2019","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","year":2019,"venue":"ACL 2019 / arXiv","authors":["Rowan Zellers","Ari Holtzman","Yonatan Bisk","Ali Farhadi","Yejin Choi"],"authors_zh":"Rowan Zellers 等（University of Washington, Allen Institute for Artificial Intelligence）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["commonsense-reasoning","adversarial-filtering","multiple-choice"],"tags":["benchmark","commonsense-reasoning","adversarial-filtering","multiple-choice"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要 commonsense 多选题基线、split 口径和 accuracy harness 审计线索的读者。","confidence":"high","one_line":["HellaSwag tests grounded commonsense continuation by adversarially filtering easy artifacts from video-caption and activity data.","HellaSwag 用约 70K 个 adversarially filtered 续写选择题测试 grounded commonsense。"],"why":"It supplies a reusable evaluation coordinate for commonsense-reasoning, adversarial-filtering work.","primary_link":"https://arxiv.org/abs/1905.07830","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/rowanz/hellaswag"},{"key":"data","label":["Data","数据"],"url":"https://github.com/rowanz/hellaswag/tree/master/data"},{"key":"project","label":["Project","项目主页"],"url":"https://rowanzellers.com/hellaswag/"}],"link_count":6,"sections":9},{"id":"pubmedqa-2019","title":"PubMedQA: A Dataset for Biomedical Research Question Answering","year":2019,"venue":"EMNLP-IJCNLP 2019 / arXiv","authors":["Qiao Jin","Bhuwan Dhingra","Zhengping Liu","William W. Cohen","Xinghua Lu"],"authors_zh":"Qiao Jin 等（University of Pittsburgh, Carnegie Mellon University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["biomedical-qa","yes-no-maybe","domain-expert-benchmark"],"tags":["benchmark","biomedical-qa","yes-no-maybe","domain-expert-benchmark"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要补齐基础 benchmark 坐标、评测契约、contamination 和 scorer 审计的读者。","confidence":"high","one_line":["PubMedQA evaluates biomedical yes/no/maybe reasoning with 1,000 expert-labeled examples plus larger unlabeled and artificial subsets.","PubMedQA 用 PubMed 摘要评测生物医学 yes/no/maybe 推理，核心评测集是 1,000 个 expert-labeled PQA-L 样本。"],"why":"It supplies a reusable evaluation coordinate for biomedical-qa, yes-no-maybe work.","primary_link":"https://aclanthology.org/D19-1259/","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/pubmedqa/pubmedqa"},{"key":"data","label":["Data","数据"],"url":"https://github.com/pubmedqa/pubmedqa/tree/master/data"},{"key":"project","label":["Project","项目主页"],"url":"https://pubmedqa.github.io"}],"link_count":6,"sections":9},{"id":"winogrande-2019","title":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale","year":2019,"venue":"AAAI 2020 / arXiv","authors":["Keisuke Sakaguchi","Ronan Le Bras","Chandra Bhagavatula","Yejin Choi"],"authors_zh":"Keisuke Sakaguchi 等（Allen Institute for Artificial Intelligence, University of Washington）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["commonsense-reasoning","winograd-schema","adversarial-filtering"],"tags":["benchmark","commonsense-reasoning","winograd-schema","adversarial-filtering"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要 Winograd-style 常识基线、subset 大小和二选一 scorer 审计线索的读者。","confidence":"high","one_line":["WinoGrande scales Winograd-style commonsense pronoun resolution with adversarial filtering to reduce superficial cues.","WinoGrande 将 Winograd-style 常识消歧扩展到约 44K 题，并用 adversarial filtering 降低浅层线索。"],"why":"It supplies a reusable evaluation coordinate for commonsense-reasoning, winograd-schema work.","primary_link":"https://arxiv.org/abs/1907.10641","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/allenai/winogrande"},{"key":"data","label":["Data","数据"],"url":"https://storage.googleapis.com/ai2-mosaic/public/winogrande/winogrande_1.1.zip"},{"key":"project","label":["Project","项目主页"],"url":"https://winogrande.allenai.org/"}],"link_count":5,"sections":9},{"id":"fever-2018","title":"FEVER: a Large-scale Dataset for Fact Extraction and VERification","year":2018,"venue":"NAACL 2018 / arXiv","authors":["James Thorne","Andreas Vlachos","Christos Christodoulopoulos","Arpit Mittal"],"authors_zh":"James Thorne 等（University of Sheffield, Amazon）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["fact-verification","evidence-retrieval","factuality"],"tags":["benchmark","fact-verification","evidence-retrieval","factuality"],"status":"verified","priority":"可读","paper_type_zh":"NAACL-HLT 2018 / arXiv 的 fact-verification benchmark","best_for_zh":"需要补齐基础 benchmark 坐标、评测契约、contamination 和 scorer 审计的读者。","confidence":"high","one_line":["FEVER links 185,445 factual claims to Wikipedia evidence and labels, making fact verification an evidence-grounded evaluation task.","FEVER 把 185,445 条 factual claims 绑定到 Wikipedia 证据句和三分类标签，是证据约束 fact verification 的经典 benchmark。"],"why":"It supplies a reusable evaluation coordinate for fact-verification, evidence-retrieval work.","primary_link":"https://arxiv.org/abs/1803.05355","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/sheffieldnlp/fever-scorer"},{"key":"data","label":["Data","数据"],"url":"https://fever.ai/dataset/fever.html"},{"key":"project","label":["Project","项目主页"],"url":"https://fever.ai/"}],"link_count":6,"sections":9},{"id":"glue-2018","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","year":2018,"venue":"ICLR 2019 / arXiv","authors":["Alex Wang","Amanpreet Singh","Julian Michael","Felix Hill","Omer Levy","Samuel R. Bowman"],"authors_zh":"Alex Wang 等（New York University, DeepMind）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["mixed"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["nlu-benchmark","sentence-understanding","benchmark-infrastructure"],"tags":["benchmark","nlu-benchmark","sentence-understanding","benchmark-infrastructure"],"status":"verified","priority":"可读","paper_type_zh":"ICLR 2019 / arXiv 的 NLU benchmark","best_for_zh":"需要 NLU 多任务基线、subtask split 口径和 leaderboard scorer 审计线索的读者。","confidence":"high","one_line":["GLUE is an ICLR 2019 multi-task NLU benchmark with nine English sentence and sentence-pair tasks plus a diagnostic set.","GLUE 是 ICLR 2019 的多任务 NLU benchmark，包含 9 个英文句子/句对任务和 diagnostic set。"],"why":"It is a foundational benchmark-suite coordinate for NLU evaluation and scorer aggregation.","primary_link":"https://arxiv.org/abs/1804.07461","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/nyu-mll/GLUE-baselines"},{"key":"data","label":["Data","数据"],"url":"https://gluebenchmark.com/"}],"link_count":4,"sections":9},{"id":"hotpotqa-2018","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","year":2018,"venue":"EMNLP 2018 / arXiv","authors":["Zhilin Yang","Peng Qi","Saizheng Zhang","Yoshua Bengio","William W. Cohen","Ruslan Salakhutdinov","Christopher D. Manning"],"authors_zh":"Zhilin Yang 等（Carnegie Mellon University, Stanford University）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["multi-hop-qa","explainable-qa","retrieval-reasoning"],"tags":["benchmark","multi-hop-qa","explainable-qa","retrieval-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要多跳检索/阅读理解评测面，或比较 distractor、full-wiki、supporting-fact scorer 的读者。","confidence":"high","one_line":["HotpotQA makes multi-hop QA auditable by pairing answers with supporting Wikipedia facts.","HotpotQA 含约 113k 个 Wikipedia 多跳问答样本，并给出 supporting facts，使答案与证据链能一起被评测。"],"why":"It supplies a reusable evaluation coordinate for multi-hop-qa, explainable-qa work.","primary_link":"https://arxiv.org/abs/1809.09600","links":[{"key":"project","label":["Project","项目主页"],"url":"https://hotpotqa.github.io/"}],"link_count":4,"sections":9},{"id":"spider-2018","title":"Spider: A Large-Scale Human-Labeled Dataset for Complex and Cross-Domain Semantic Parsing and Text-to-SQL Task","year":2018,"venue":"EMNLP 2018 / arXiv","authors":["Tao Yu","Rui Zhang","Kai Yang","Michihiro Yasunaga","Dongxu Wang","Zifan Li","James Ma","Irene Li","Qingning Yao","Shanelle Roman","Zilin Zhang","Dragomir Radev"],"authors_zh":"Tao Yu 等（Yale University, Salesforce Research）","tracks":["benchmarks_evaluation_surfaces"],"source_role":["benchmark"],"verification_contract":["programmatic"],"supervision_granularity":["answer_level"],"training_use":["evaluation","audit"],"construction_layer":["prompt_sourcing","release_audit"],"domains":["text-to-sql","semantic-parsing","database-reasoning"],"tags":["benchmark","text-to-sql","semantic-parsing","database-reasoning"],"status":"verified","priority":"可读","paper_type_zh":"Benchmark / evaluation surface","best_for_zh":"需要 text-to-SQL schema 泛化、SQL scorer 和数据库版本审计线索的读者。","confidence":"high","one_line":["Spider made cross-domain text-to-SQL evaluation difficult by separating database schemas across train and test.","Spider 用 10,181 个问题、200 个数据库和跨 domain split 建立了经典 text-to-SQL 难题。"],"why":"It supplies a reusable evaluation coordinate for text-to-sql, semantic-parsing work.","primary_link":"https://arxiv.org/abs/1809.08887","links":[{"key":"code","label":["Code","代码"],"url":"https://github.com/taoyds/spider"},{"key":"data","label":["Data","数据"],"url":"https://yale-lily.github.io/spider"}],"link_count":4,"sections":9}]
