{"api_version":"v1","generated_at":"2026-09-29T13:00:45","count":2794,"scope":"judged","fields":["id","title","zh_title","primary_category","date","score","bucket","tags","rubric_hits","abs_url","has_summary"],"papers":[{"id":"2609.30563","title":"Thinking Less to Simulate Better: Intuitive Prompting Improves LLM Agents Simulating Individual Social Media Reactions, Including Unfamiliar Content","zh_title":"少思考以更好地模拟：直觉提示提升LLM智能体模拟个体社交媒体反应（包括不熟悉内容）","primary_category":"cs.AI","date":"2026-09-28","score":10,"bucket":"selected","tags":["LLM仿真","人类行为对照","提示策略"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.30563","has_summary":true},{"id":"2609.30883","title":"Warned alike, AI agents avoid the less-crowded road while people take it","zh_title":"同样被警告，AI智能体避开较不拥挤的道路而人类选择它","primary_category":"physics.soc-ph","date":"2026-09-28","score":10,"bucket":"selected","tags":["LLM仿真","拥堵博弈","人机对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.30883","has_summary":true},{"id":"2609.30896","title":"Large language models underestimate and partly misrepresent cultural variation in everyday norms","zh_title":"大语言模型低估并部分误现日常规范的文化差异","primary_category":"cs.CY","date":"2026-09-28","score":10,"bucket":"selected","tags":["文化规范","仿真偏差","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.30896","has_summary":true},{"id":"2604.20050","title":"Information Aggregation with AI Agents","zh_title":"AI代理的信息聚合研究","primary_category":"econ.GN","date":"2026-09-28","score":9,"bucket":"selected","tags":["LLM仿真","信息聚合","行为实验"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.20050","has_summary":true},{"id":"2609.02729","title":"BuildOcc: A Large Language Model Occupant Agent Platform for Building Energy Research","zh_title":"BuildOcc：用于建筑能源研究的大语言模型居住者智能体平台","primary_category":"cs.HC","date":"2026-09-28","score":9,"bucket":"selected","tags":["LLM仿真","建筑能源","ATUS数据"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.02729","has_summary":true},{"id":"2609.30940","title":"Financial Fragility in Societies of LLM Agents: Coordination Failures and Stabilizing Mechanisms","zh_title":"LLM智能体社会中的金融脆弱性：协调失败与稳定机制","primary_category":"cs.AI","date":"2026-09-28","score":8,"bucket":"selected","tags":["LLM智能体","金融仿真","协调失败"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.30940","has_summary":true},{"id":"2609.31054","title":"Cheap, open agents make LLM pollution harder to mitigate","zh_title":"廉价开放智能体使LLM污染更难缓解","primary_category":"cs.AI","date":"2026-09-28","score":8,"bucket":"selected","tags":["LLM污染","调查数据","检测方法"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.31054","has_summary":true},{"id":"2608.27167","title":"Calibrated Enough to Know, Not Calibrated to Act: Fabricated Evidence Makes LLM Agents Commit to the Unknowable","zh_title":"校准到知道，但未校准到行动：伪造证据使LLM智能体对不可知问题做出承诺","primary_category":"cs.AI","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM决策偏差","可靠性评估","批判性研究"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.27167","has_summary":true},{"id":"2609.30867","title":"Evidence-Grounded Auditing of Identification Assumptions in Climate-Policy Causal Evaluations","zh_title":"气候政策因果评估中识别假设的证据基础审计","primary_category":"cs.CL","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM审计","因果推断","方法论"],"rubric_hits":["A4","B3"],"abs_url":"https://arxiv.org/abs/2609.30867","has_summary":true},{"id":"2609.31245","title":"RupeeBias: Auditing Demographic Bias in Indian Economic Guidance from Large Language Models","zh_title":"RupeeBias：审计印度经济指导中大语言模型的人口统计偏差","primary_category":"cs.CL","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM偏差审计","经济决策仿真","人口统计代表性"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.31245","has_summary":true},{"id":"2609.31013","title":"Same Text, Different Numbers: The Divergence of LLM-Based Measures","zh_title":"相同文本，不同数字：基于LLM的测量分歧","primary_category":"cs.AI","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM测量一致性","算法保真度","文本分析"],"rubric_hits":["A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2609.31013","has_summary":true},{"id":"2609.31468","title":"PriceBench: A Diagnostic Benchmark for Price, Quality, and Brand Preferences in LLM Booking Agents","zh_title":"PriceBench：LLM预订代理中价格、质量与品牌偏好的诊断基准","primary_category":"econ.GN","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM仿真","消费者选择","偏好测量"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.31468","has_summary":true},{"id":"2609.30705","title":"The Price of Thought: Does Test-Time Reasoning Pay in LLM Trading?","zh_title":"思考的代价：测试时推理在LLM交易中是否值得？","primary_category":"cs.AI","date":"2026-09-28","score":7,"bucket":"pending","tags":["LLM交易","推理成本","经济决策"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.30705","has_summary":true},{"id":"2609.31095","title":"Confident, Not Wiser: The Dunning-Kruger Effect in Human-AI Interaction","zh_title":"自信而非更明智：人机交互中的达克效应","primary_category":"cs.HC","date":"2026-09-28","score":7,"bucket":"pending","tags":["人机交互","元认知","AI辅助决策"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.31095","has_summary":true},{"id":"2609.31046","title":"Modeling Student Sensemaking with LLMs and Knowledge-Graph-Guided Inference","zh_title":"用大语言模型和知识图谱引导推理建模学生意义建构","primary_category":"cs.CL","date":"2026-09-28","score":6,"bucket":"other","tags":["LLM标注","教育对话分析","知识图谱"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.31046","has_summary":false},{"id":"2609.31078","title":"OmouAI: Argumentative Human-AI Policy Deliberation with Simulated Personas","zh_title":"OmouAI：基于模拟人格的论证式人机政策协商","primary_category":"cs.AI","date":"2026-09-28","score":6,"bucket":"other","tags":["LLM仿真","政策协商","计算论证"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.31078","has_summary":false},{"id":"2609.31607","title":"Statistical attribute alignment for black-box generative AI via output post-processing","zh_title":"通过输出后处理实现黑盒生成式AI的统计属性对齐","primary_category":"stat.ME","date":"2026-09-28","score":6,"bucket":"other","tags":["生成式AI","属性对齐","合成数据"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.31607","has_summary":false},{"id":"2609.30716","title":"Words Speak Louder Than Order: A Behavioral Evaluation of Gemma 4","zh_title":"言语胜于顺序：Gemma 4 的行为评估","primary_category":"cs.CL","date":"2026-09-28","score":5,"bucket":"other","tags":["LLM行为评估","来源框架","顺序效应"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.30716","has_summary":false},{"id":"2609.30986","title":"Evaluating Sycophancy in Chinese Large Language Models on Factual Questions Derived from Online Search Queries","zh_title":"评估中文大语言模型在基于在线搜索查询的事实问题上的谄媚行为","primary_category":"cs.CL","date":"2026-09-28","score":5,"bucket":"other","tags":["LLM谄媚","事实准确性","中文模型"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.30986","has_summary":false},{"id":"2609.31506","title":"Evaluating Cultural Awareness of LLMs for Haitian Creole","zh_title":"评估LLM对海地克里奥尔语的文化意识","primary_category":"cs.CL","date":"2026-09-28","score":5,"bucket":"other","tags":["文化意识","低资源语言","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.31506","has_summary":false},{"id":"2609.31603","title":"User Model Extraction via Belief Self-Distillation","zh_title":"通过信念自蒸馏提取用户模型","primary_category":"cs.LG","date":"2026-09-28","score":5,"bucket":"other","tags":["LLM内部表征","用户建模","可解释性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.31603","has_summary":false},{"id":"2609.31260","title":"Agentic Limit Order Books: Phase Transitions and Market Impact","zh_title":"智能体限价订单簿：相变与市场冲击","primary_category":"q-fin.TR","date":"2026-09-28","score":5,"bucket":"other","tags":["多智能体市场模拟","限价订单簿","强化学习"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.31260","has_summary":false},{"id":"2608.29420","title":"One Capability or Many? Structural and Predictive Tests of Benchmark Validity Disagree About Economic Benchmarks for Frontier AI","zh_title":"一种能力还是多种？基准效度的结构与预测检验对前沿AI经济基准意见不一","primary_category":"cs.LG","date":"2026-09-28","score":2,"bucket":"other","tags":["基准评估","构念效度","AI评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29420","has_summary":false},{"id":"2609.28504","title":"AI in Science: Early Insights","zh_title":"AI在科学中的早期洞察","primary_category":"econ.GN","date":"2026-09-28","score":2,"bucket":"other","tags":["AI与科学","生产力","科学过程"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.28504","has_summary":false},{"id":"2609.30492","title":"Breaking Homogeneity: Diversifying Persona Sets for Creative LLM Outputs","zh_title":"打破同质性：为创造性LLM输出多样化角色集","primary_category":"cs.CL","date":"2026-09-28","score":2,"bucket":"other","tags":["角色多样化","创造力生成","提示工程"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.30492","has_summary":false},{"id":"2609.30558","title":"Probing Stability-Plasticity Tradeoffs in Agent Memory through Cognitive Experimental Paradigms","zh_title":"通过认知实验范式探究智能体记忆中的稳定性-可塑性权衡","primary_category":"cs.CL","date":"2026-09-28","score":2,"bucket":"other","tags":["智能体记忆","认知诊断","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.30558","has_summary":false},{"id":"2609.30897","title":"From annotation to reasoning: Culture in language models","zh_title":"从标注到推理：语言模型中的文化","primary_category":"cs.CL","date":"2026-09-28","score":2,"bucket":"other","tags":["文化基准","文学解释","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30897","has_summary":false},{"id":"2609.30768","title":"Does Thinking Help Fairness? Reasoning Tokens Resolve Some Biases but Create More","zh_title":"思考有助于公平吗？推理令牌解决一些偏差但创造更多","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["公平性","推理模型","偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30768","has_summary":false},{"id":"2609.30939","title":"MACBT: A Multi-Agent Cognitive Behavioral Therapy Decision Support System with Longitudinal Memory","zh_title":"MACBT：具有纵向记忆的多智能体认知行为疗法决策支持系统","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["多智能体系统","认知行为疗法","决策支持"],"rubric_hits":["C1","C3"],"abs_url":"https://arxiv.org/abs/2609.30939","has_summary":false},{"id":"2609.31184","title":"Accounting for Bias Enables Sustainable LLM Evaluation","zh_title":"考虑偏差实现可持续的LLM评估","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["LLM评估","偏差校正","测量模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.31184","has_summary":false},{"id":"2609.31215","title":"DIAL: Position-Debiased LLM Judges with Adaptive Human Preference Calibration","zh_title":"DIAL：基于自适应人类偏好校准的位置去偏LLM评判器","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["LLM评判","偏好校准","评估方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.31215","has_summary":false},{"id":"2609.31473","title":"Game Arena: Strategic LLM Evaluation in Competitive Environments","zh_title":"游戏竞技场：竞争环境中的大语言模型战略评估","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["LLM评估","游戏AI","多智能体"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.31473","has_summary":false},{"id":"2609.31563","title":"Multi-agent Scaling Across Disjunctive and Compensatory Tasks","zh_title":"分离型与补偿型任务中的多智能体扩展","primary_category":"cs.AI","date":"2026-09-28","score":2,"bucket":"other","tags":["多智能体系统","任务扩展","LLM协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.31563","has_summary":false},{"id":"2609.30466","title":"A Benchmarking Framework for Context-aware XR Interfaces","zh_title":"面向情境感知XR界面的基准测试框架","primary_category":"cs.HC","date":"2026-09-28","score":2,"bucket":"other","tags":["XR界面","基准测试","LLM方法"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.30466","has_summary":false},{"id":"2609.31219","title":"Research with AI Agents: How Agentic Systems Are Changing Scientific Work","zh_title":"AI代理研究：代理系统如何改变科学工作","primary_category":"cs.CY","date":"2026-09-28","score":2,"bucket":"other","tags":["AI代理","科研自动化","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.31219","has_summary":false},{"id":"2609.30427","title":"Fake News Theories: Harnessing Disciplinary Insights for Computational Modeling, Detection, and Explanation","zh_title":"假新闻理论：利用跨学科洞见进行计算建模、检测与解释","primary_category":"cs.LG","date":"2026-09-28","score":2,"bucket":"other","tags":["假新闻检测","可解释性","跨学科理论"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30427","has_summary":false},{"id":"2609.31590","title":"AgentWorld: Benchmarking Long-Horizon Collaboration of Multi-agent LLMs","zh_title":"AgentWorld：多智能体大语言模型长程协作基准测试","primary_category":"cs.MA","date":"2026-09-28","score":2,"bucket":"other","tags":["多智能体协作","基准测试","任务完成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.31590","has_summary":false},{"id":"2609.30405","title":"Adaptive Multi-Value Control in LLMs via Causal Activation Steering","zh_title":"通过因果激活引导实现LLM的自适应多值控制","primary_category":"cs.LG","date":"2026-09-28","score":2,"bucket":"other","tags":["激活引导","价值观对齐","模型控制"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.30405","has_summary":false},{"id":"2609.06951","title":"Steering Interference Reflects the Model's Defaults, Not the Behavior Directions","zh_title":"引导干扰反映模型默认行为而非行为方向","primary_category":"cs.LG","date":"2026-09-28","score":0,"bucket":"other","tags":["激活引导","模型可解释性","行为控制"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.06951","has_summary":false},{"id":"2609.29390","title":"Likelihood Ranking doesn't Scale Like Prompting in LLMs","zh_title":"似然排序在LLM中不像提示那样随规模扩展","primary_category":"cs.CL","date":"2026-09-28","score":0,"bucket":"other","tags":["LLM评测","多项选择问答","似然排序"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.29390","has_summary":false},{"id":"2609.30849","title":"Enhancing Assessment of Self-Consistency in LLM Explanations using Perturbation Strength","zh_title":"利用扰动强度增强对LLM解释自洽性的评估","primary_category":"cs.CL","date":"2026-09-28","score":0,"bucket":"other","tags":["LLM评估","解释自洽性","扰动方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30849","has_summary":false},{"id":"2609.31166","title":"AgentRecommender: LLM Agents Enable Customizable Recommender Systems on the User Side","zh_title":"AgentRecommender：LLM智能体实现用户侧可定制推荐系统","primary_category":"cs.IR","date":"2026-09-28","score":0,"bucket":"other","tags":["推荐系统","LLM智能体","用户侧"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.31166","has_summary":false},{"id":"2609.31272","title":"Cognitive Skills in the Age of AI: Computing Students and Experts Perceptions","zh_title":"AI时代的认知技能：计算专业学生与专家的看法","primary_category":"cs.HC","date":"2026-09-28","score":0,"bucket":"other","tags":["认知技能","AI影响","人机交互"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.31272","has_summary":false},{"id":"2609.30388","title":"The Interviewer's Perspective: Unpacking the Impact of Real-Time AI Interviewing Assistance on Social Dynamics","zh_title":"访谈者视角：实时AI访谈辅助对社会动态的影响","primary_category":"cs.HC","date":"2026-09-28","score":0,"bucket":"other","tags":["AI辅助访谈","人机交互","社会动态"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.30388","has_summary":false},{"id":"2609.30588","title":"Orchestrating GenAI for Interdisciplinary Research","zh_title":"为跨学科研究编排生成式人工智能","primary_category":"cs.HC","date":"2026-09-28","score":0,"bucket":"other","tags":["人机交互","跨学科研究","GenAI使用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.30588","has_summary":false},{"id":"2609.31060","title":"The Crowd in the Machine: A Crisis-Informatics Reading of the 2026 Autonomous Agent Incidents","zh_title":"机器中的群体：对2026年自主智能体事件的危机信息学解读","primary_category":"cs.MA","date":"2026-09-28","score":0,"bucket":"other","tags":["多智能体系统","危机信息学","涌现行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.31060","has_summary":false},{"id":"2609.27690","title":"Consequential Behaviour and Representational Fairness in the Validation of Synthetic Research","zh_title":"合成研究验证中的后果行为与表征公平性","primary_category":"cs.CL","date":"2026-09-25","score":10,"bucket":"selected","tags":["LLM仿真","验证框架","公平性"],"rubric_hits":["A1","A2","A4","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.27690","has_summary":true},{"id":"2609.29928","title":"Cultural Divergence Preservation: Diagnosing Flattening and Caricature in LLM-Simulated Survey Populations","zh_title":"文化差异保持：诊断LLM模拟调查人群中的扁平化与夸张化","primary_category":"cs.CL","date":"2026-09-25","score":10,"bucket":"selected","tags":["LLM仿真","跨文化调查","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.29928","has_summary":true},{"id":"2609.30030","title":"Artificial Societies Benchmark: A Validation Framework for Synthetic Research","zh_title":"人工社会基准：合成研究的验证框架","primary_category":"cs.CL","date":"2026-09-25","score":10,"bucket":"selected","tags":["LLM仿真","效度验证","合成人群"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.30030","has_summary":true},{"id":"2609.29952","title":"Augur: A Synthetic Decision Lab for Rehearsing Reactions to Product and Policy Changes","zh_title":"Augur：用于预演产品和政策变化反应的合成决策实验室","primary_category":"cs.AI","date":"2026-09-25","score":9,"bucket":"selected","tags":["LLM仿真","人类行为预测","政策评估"],"rubric_hits":["A1","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.29952","has_summary":true},{"id":"2609.29692","title":"Fair Like Us? Auditing LLM Alignment in Resource Allocation","zh_title":"像我们一样公平？审计资源分配中LLM的对齐","primary_category":"cs.AI","date":"2026-09-25","score":9,"bucket":"selected","tags":["LLM仿真","公平分配","人类对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.29692","has_summary":true},{"id":"2609.29143","title":"AI-Moderated Interviews for Market Research and Digital Twins Calibration","zh_title":"用于市场研究和数字孪生校准的AI主持访谈","primary_category":"cs.CY","date":"2026-09-25","score":9,"bucket":"selected","tags":["LLM仿真","数字孪生","市场研究"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.29143","has_summary":true},{"id":"2609.29370","title":"From Policy Documents to Structured Survey Responses: Evaluating Large Language Models for Policy Monitoring","zh_title":"从政策文件到结构化调查回答：评估大语言模型用于政策监测","primary_category":"cs.CL","date":"2026-09-25","score":8,"bucket":"selected","tags":["LLM仿真","政策监测","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.29370","has_summary":true},{"id":"2609.28486","title":"Political Sorting Can Drive AI Models Apart Through User Feedback","zh_title":"政治分类可通过用户反馈使AI模型分化","primary_category":"cs.CY","date":"2026-09-25","score":8,"bucket":"selected","tags":["LLM仿真","政治极化","人类对照"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.28486","has_summary":true},{"id":"2609.22904","title":"LLMs Anchor on Chief Complaint and Fail to Integrate Evidence in Sequential Clinical Triage","zh_title":"LLM在顺序临床分诊中锚定主诉且未能整合证据","primary_category":"cs.CL","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM仿真","临床决策","可靠性评估"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.22904","has_summary":true},{"id":"2609.28673","title":"Benchmarking Argumentative Behaviour of LLMs: A Study of Defences Against Character Attacks","zh_title":"基准测试大语言模型的论辩行为：对人身攻击防御策略的研究","primary_category":"cs.CL","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM仿真","论辩行为","政治辩论"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.28673","has_summary":true},{"id":"2609.30137","title":"Screen Before You Serve: Simulation for Production Customer Experience AI Agents at 140M Scale","zh_title":"先模拟后上线：1.4亿规模生产客户体验AI智能体的仿真","primary_category":"cs.AI","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM仿真","客户服务","A/B测试"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.30137","has_summary":true},{"id":"2609.28690","title":"Beyond Surface Style: Aligning Multi-Turn User Simulators with Behavioral Consistency","zh_title":"超越表面风格：对齐多轮用户模拟器的行为一致性","primary_category":"cs.AI","date":"2026-09-25","score":7,"bucket":"pending","tags":["用户模拟","强化学习","行为对齐"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.28690","has_summary":true},{"id":"2609.28876","title":"Forecast-Dojo: Replayable Environments for Benchmarking and Training LLM Forecasting Agents","zh_title":"Forecast-Dojo：用于基准测试和训练LLM预测代理的可重放环境","primary_category":"cs.AI","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM预测代理","预测市场","人类行为对照"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.28876","has_summary":true},{"id":"2609.28820","title":"AI-Enabled Human Memory Manipulation: Misleading AI-Generated Summaries Distort Human Memory","zh_title":"AI赋能的人类记忆操纵：误导性AI生成摘要扭曲人类记忆","primary_category":"cs.CY","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM误导信息","人类记忆","人机交互实验"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.28820","has_summary":true},{"id":"2609.29513","title":"Signed Exposure: Fair Routing of Algorithmic Attention When Attention Can Harm","zh_title":"符号化曝光：当注意力可能造成伤害时算法注意力的公平路由","primary_category":"cs.CY","date":"2026-09-25","score":7,"bucket":"pending","tags":["LLM仿真","公平路由","人类数据对照"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.29513","has_summary":true},{"id":"2609.29001","title":"Polite but Misaligned: Evaluating LLM Politeness Judgments Against Human Pragmatic Norms","zh_title":"礼貌但错位：评估大语言模型礼貌判断与人类语用规范的一致性","primary_category":"cs.CL","date":"2026-09-25","score":6,"bucket":"other","tags":["LLM评估","语用规范","人机对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.29001","has_summary":false},{"id":"2609.29508","title":"Evaluation of Multi-Turn Consistency in LLM Agents: Survival Analysis and Failure-Rationale Taxonomy","zh_title":"LLM智能体多轮一致性评估：生存分析与失败理由分类","primary_category":"cs.AI","date":"2026-09-25","score":6,"bucket":"other","tags":["LLM智能体","社会模拟","一致性评估"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.29508","has_summary":false},{"id":"2609.29509","title":"Delay-of-Gratification as a Multi-Agent Survival Micro-benchmark for Long-Horizon LLMs: Social Exposure, Personas, and Tool Use Budgets","zh_title":"延迟满足作为长时程LLM的多智能体生存微基准：社会暴露、人设与工具使用预算","primary_category":"cs.AI","date":"2026-09-25","score":6,"bucket":"other","tags":["LLM仿真","多智能体","延迟满足"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.29509","has_summary":false},{"id":"2609.28547","title":"PAWS: Policy-driven Agentic World Simulation","zh_title":"PAWS：政策驱动的智能体世界仿真","primary_category":"cs.AI","date":"2026-09-25","score":6,"bucket":"other","tags":["多智能体仿真","金融政策","数据集"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.28547","has_summary":false},{"id":"2609.30028","title":"How does Adversarial Influence Scale in Multi-Agent Systems?","zh_title":"多智能体系统中的对抗性影响如何随规模变化？","primary_category":"cs.AI","date":"2026-09-25","score":6,"bucket":"other","tags":["多智能体系统","社会影响","LLM仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.30028","has_summary":false},{"id":"2609.11144","title":"Human Agreement and Return Association Are Not Interchangeable Criteria","zh_title":"人类一致性与收益关联并非可互换的准则","primary_category":"cs.AI","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM标注","金融NLP","效度评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.11144","has_summary":false},{"id":"2609.21277","title":"How Many Humans Are 32 LLM Judges Worth?","zh_title":"32个LLM法官相当于多少人类？","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM标注","人类等效","标注可靠性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.21277","has_summary":true},{"id":"2609.25447","title":"Conduct Under Pressure: What Sixty Language Models Do When a User Pushes","zh_title":"压力下的行为：六十个语言模型在用户施压时会做什么","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM行为","模型评估","人机交互"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.25447","has_summary":false},{"id":"2609.28487","title":"Framing by Wording, Framing by Selection: A Large-Scale Two-Dimensional Audit of French News Headlines, 2022-2025","zh_title":"措辞框架与选择框架：法国新闻标题的大规模二维审计（2022-2025）","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM标注","框架分析","新闻标题"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.28487","has_summary":false},{"id":"2609.29333","title":"Where LLM Graders Succeed and Break: Evidence from Two Computer-Science Exams","zh_title":"LLM评分器在哪里成功与失败：来自两场计算机科学考试的证据","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM评分","教育评估","可靠性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.29333","has_summary":false},{"id":"2609.29807","title":"CORDIAL: Calibrating Ordinal LLM Outputs from Few Labels","zh_title":"CORDIAL：从少量标签校准序数型LLM输出","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM校准","序数输出","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.29807","has_summary":false},{"id":"2609.30012","title":"Low-Cost Assays for Measuring Model Behavior Across Vendors and Releases","zh_title":"跨供应商和版本测量模型行为的低成本方法","primary_category":"cs.CL","date":"2026-09-25","score":5,"bucket":"other","tags":["模型行为测量","LLM评估","跨模型比较"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.30012","has_summary":false},{"id":"2609.28859","title":"Human-AI-Powered Hypothesis Testing: Cost-Aware Selective AI Scoring and Sequential Human Escalation","zh_title":"人机协同的假设检验：成本感知的选择性AI评分与序贯人工升级","primary_category":"cs.AI","date":"2026-09-25","score":5,"bucket":"other","tags":["AI标注","假设检验","成本优化"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.28859","has_summary":false},{"id":"2609.29431","title":"Calibrating LLM Judges for Human and AI Conversations","zh_title":"校准用于人类与AI对话的LLM评判者","primary_category":"cs.HC","date":"2026-09-25","score":5,"bucket":"other","tags":["LLM评判","对话质量","校准"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.29431","has_summary":false},{"id":"2609.28483","title":"Generative AI May Reinforce Social Biases in Software Engineering Education","zh_title":"生成式AI可能强化软件工程教育中的社会偏见","primary_category":"cs.CY","date":"2026-09-25","score":5,"bucket":"other","tags":["生成式AI","社会偏见","教育"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.28483","has_summary":false},{"id":"2609.29701","title":"Multi-Agent Debate for Explainable Trading: Reasoning, Consensus, and Performance in Simulated Markets","zh_title":"面向可解释交易的多智能体辩论：模拟市场中的推理、共识与表现","primary_category":"cs.MA","date":"2026-09-25","score":5,"bucket":"other","tags":["多智能体辩论","金融市场模拟","LLM决策"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.29701","has_summary":false},{"id":"2609.29958","title":"Multi-Dimensional Matching","zh_title":"多维匹配","primary_category":"econ.EM","date":"2026-09-25","score":3,"bucket":"other","tags":["匹配机制","算法设计","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29958","has_summary":false},{"id":"2607.25021","title":"Chart-Supported or Model-Supplied? Examining MLLM-Generated Claims for Accessible Visualization","zh_title":"图表支持还是模型提供？考察MLLM生成声明以实现可访问可视化","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["多模态大模型","可视化描述","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.25021","has_summary":false},{"id":"2609.05059","title":"Measuring Brand and Source Discovery under Repeated LLM Queries: A Finite-Sample Audit","zh_title":"重复LLM查询下品牌与来源发现的测量：有限样本审计","primary_category":"cs.IR","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM审计","信息检索","有限样本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.05059","has_summary":false},{"id":"2609.11489","title":"The Convention Gap: Towards Measuring Implicit Communication in Cooperative AI Evaluation","zh_title":"惯例差距：迈向合作AI评估中的隐式沟通测量","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["合作AI","隐式沟通","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11489","has_summary":false},{"id":"2609.13422","title":"Vibe Patenting: Evaluating LLM Judges for Professional Patent-Drafting Agents","zh_title":"氛围专利撰写：评估专业专利起草代理中的LLM法官","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM评估","多智能体系统","专利撰写"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.13422","has_summary":false},{"id":"2609.15494","title":"The Troy Moment: How LLM Agents Adjudicate the Decision Point Under Impossible Tasks, Claimed Authority, and Peer Information","zh_title":"特洛伊时刻：LLM智能体如何在不可能任务、声称权威与同伴信息下裁决决策点","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM智能体","任务失败","对齐"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15494","has_summary":false},{"id":"2609.26758","title":"Type-Safe Is Not Error-Free: A Constrained Decision Head Follows the Option Name, Not the Rubric Bound to It","zh_title":"类型安全并非无错：约束决策头跟随选项名称而非其绑定的规则","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM决策鲁棒性","选项名称效应","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26758","has_summary":false},{"id":"2609.29418","title":"Controlling Backchannels in Streamable Full-duplex Models","zh_title":"在可流式全双工模型中控制反馈通道","primary_category":"cs.CL","date":"2026-09-25","score":2,"bucket":"other","tags":["对话系统","反馈通道","全双工模型"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29418","has_summary":false},{"id":"2609.29445","title":"Two Emojis of Difference: What Multilingual Affective Generation Benchmarks Actually Measure","zh_title":"两个表情符号的差异：多语言情感生成基准实际测量了什么","primary_category":"cs.CL","date":"2026-09-25","score":2,"bucket":"other","tags":["基准审计","情感生成","评测偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.29445","has_summary":false},{"id":"2609.29494","title":"Who Put the I in AI? Provenance and the Admissibility of Machine Self-Report","zh_title":"谁把‘我’放进了AI？机器自我报告的来源与可采性","primary_category":"cs.CL","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM自我报告","认识论","训练溯源"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29494","has_summary":false},{"id":"2609.29429","title":"Just Ask Jev: Reinforcement Learning for Calibrated Decisions as a Zero-Shot Detector of AI Alignment Failures","zh_title":"只需问Jev：作为AI对齐失败零样本检测器的校准决策强化学习","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["AI对齐","基准评测","检测器"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.29429","has_summary":false},{"id":"2609.29528","title":"A Corpus of Real Scam- and Spam-Call Conversations from an Active Voice-Agent Honeypot","zh_title":"来自主动语音代理蜜罐的真实诈骗与垃圾电话对话语料库","primary_category":"cs.CR","date":"2026-09-25","score":2,"bucket":"other","tags":["语音代理","诈骗电话","数据集"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29528","has_summary":false},{"id":"2609.29709","title":"Three Ways Classical Test Theory Misleads for LLM Judges","zh_title":"经典测验理论误导LLM评判者的三种方式","primary_category":"cs.LG","date":"2026-09-25","score":2,"bucket":"other","tags":["LLM评估","测量理论","信度"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.29709","has_summary":false},{"id":"2609.28609","title":"Adversarial Closed-Loop Curriculum for Evolving Role-Playing Agents","zh_title":"用于进化角色扮演智能体的对抗式闭环课程","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["角色扮演智能体","强化学习","课程学习"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.28609","has_summary":false},{"id":"2609.28692","title":"Driving Epidemic Models with AI Agents: the Epydemix Agent Framework","zh_title":"用 AI 智能体驱动流行病模型：Epydemix Agent 框架","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["AI agent","流行病建模","工具调用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28692","has_summary":false},{"id":"2609.28771","title":"Agent Memory with Episodic Retrieval for Financial Decision-Making","zh_title":"基于情景检索的智能体记忆用于金融决策","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["多智能体系统","金融交易","记忆增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28771","has_summary":false},{"id":"2609.28942","title":"From Static Personal Values to Contextualized Personalization: Bayesian Personalized Value Alignment for LLMs","zh_title":"从静态个人价值观到情境化个性化：面向大语言模型的贝叶斯个性化价值对齐","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["个性化对齐","价值对齐","贝叶斯方法"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.28942","has_summary":false},{"id":"2609.29366","title":"Epistemic-Probabilistic Model for Guarded Multi-Agent LLM Coordination","zh_title":"用于受保护多智能体LLM协调的认知概率模型","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["多智能体系统","协调机制","神经符号架构"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29366","has_summary":false},{"id":"2609.29730","title":"The Gold in Bias: Maturing the AI Design Process through Verification","zh_title":"偏差中的黄金：通过验证成熟AI设计过程","primary_category":"cs.AI","date":"2026-09-25","score":2,"bucket":"other","tags":["AI偏差","验证框架","AI治理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.29730","has_summary":false},{"id":"2609.29993","title":"Will It Teach as Intended? How Teachers Configure Educational AI Chatbots","zh_title":"它会按预期教学吗？教师如何配置教育AI聊天机器人","primary_category":"cs.HC","date":"2026-09-25","score":2,"bucket":"other","tags":["教育聊天机器人","教师配置","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29993","has_summary":false},{"id":"2609.28537","title":"Privacy Leakage Through AI-mediated Analysis of Smartphone Data","zh_title":"通过AI介导的智能手机数据分析导致的隐私泄露","primary_category":"cs.CR","date":"2026-09-25","score":2,"bucket":"other","tags":["隐私泄露","LLM推断","用户研究"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.28537","has_summary":false},{"id":"2606.30583","title":"The Cross-Section of Stock Returns and AI Exposure","zh_title":"股票收益横截面与AI暴露","primary_category":"cs.CY","date":"2026-09-25","score":0,"bucket":"other","tags":["金融经济学","AI暴露","资产定价"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2606.30583","has_summary":false},{"id":"2608.22631","title":"Learning Generalizable Behaviors for Terminal Agents","zh_title":"学习终端智能体的可泛化行为","primary_category":"cs.LG","date":"2026-09-25","score":0,"bucket":"other","tags":["终端智能体","强化学习","泛化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22631","has_summary":false},{"id":"2609.14770","title":"How broad is that claim? Mapping Generalisation in NLP Research","zh_title":"这个说法有多宽泛？映射NLP研究中的泛化","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["科学声明分析","NLP元研究","文本分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.14770","has_summary":false},{"id":"2609.18366","title":"Bad Genius: Counterfactual-Guided Harness Evolution Beyond Task-Specific Shortcuts","zh_title":"坏天才：超越任务特定捷径的反事实引导的测试框架进化","primary_category":"cs.AI","date":"2026-09-25","score":0,"bucket":"other","tags":["智能体评估","基准优化","反事实搜索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18366","has_summary":false},{"id":"2609.26865","title":"Safety Nudges: User-Facing Interventions for Real-Time AI Risk Awareness","zh_title":"安全提示：面向用户的实时AI风险意识干预","primary_category":"cs.HC","date":"2026-09-25","score":0,"bucket":"other","tags":["AI安全","人机交互","用户研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.26865","has_summary":false},{"id":"2609.27265","title":"What fidelity metrics miss: a structural check on synthetic educational data","zh_title":"保真度指标遗漏了什么：对合成教育数据的结构性检查","primary_category":"cs.CY","date":"2026-09-25","score":0,"bucket":"other","tags":["差分隐私","合成数据","教育数据"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.27265","has_summary":false},{"id":"2609.29410","title":"Large Language Models for Programming: Actually Fixing or Reimplementing Incorrect Code?","zh_title":"用于编程的大型语言模型：实际修复还是重新实现错误代码？","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["代码修复","LLM编程","软件工程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29410","has_summary":false},{"id":"2609.29657","title":"How To Do Things With Prompts","zh_title":"如何用提示词做事","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["提示词","语用学","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29657","has_summary":false},{"id":"2609.29672","title":"LLMersion: A Local-First AI Agent Framework for Low-Cost Home Language Learning toward Educational Equity","zh_title":"LLMersion：面向教育公平的低成本家庭语言学习的本地优先AI智能体框架","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["AI教育","语言学习","开源框架"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29672","has_summary":false},{"id":"2609.30074","title":"How Reproducible Are Evaluation Conclusions? A Self-Audit of LLM-Inferred Prompt Structure","zh_title":"评估结论的可复现性如何？基于LLM推断提示结构的自我审计","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["LLM评测","可复现性","提示结构推断"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30074","has_summary":false},{"id":"2609.30151","title":"Does a model's stated reason for rejecting a candidate do any work?","zh_title":"模型拒绝候选者时陈述的理由是否起作用？","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["可解释性","因果推断","语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.30151","has_summary":false},{"id":"2609.30250","title":"Agentic Detection of Online Conspiracies","zh_title":"在线阴谋论的智能体检测","primary_category":"cs.CL","date":"2026-09-25","score":0,"bucket":"other","tags":["阴谋论检测","多智能体系统","社交媒体分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.30250","has_summary":false},{"id":"2609.28850","title":"RECLAIM: Can Agents Reproduce the Claims of Machine Learning Papers?","zh_title":"RECLAIM：智能体能复现机器学习论文的声明吗？","primary_category":"cs.AI","date":"2026-09-25","score":0,"bucket":"other","tags":["AI代理","论文复现","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28850","has_summary":false},{"id":"2609.29345","title":"The Last Human Gate: Forward Deployed Engineering for Governance Automation","zh_title":"最后的人类关卡：面向治理自动化的前沿部署工程","primary_category":"cs.AI","date":"2026-09-25","score":0,"bucket":"other","tags":["治理自动化","任务替代","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29345","has_summary":false},{"id":"2609.29381","title":"An auditable conditional-strategy framework for open-ended decision-making in complex lung cancer","zh_title":"复杂肺癌开放式决策的可审计条件策略框架","primary_category":"cs.AI","date":"2026-09-25","score":0,"bucket":"other","tags":["临床决策支持","LLM辅助","肺癌治疗"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.29381","has_summary":false},{"id":"2609.28737","title":"Policy Complexity, Reaction Time, and Bounded Rationality in Reinforcement Learning","zh_title":"强化学习中的策略复杂性、反应时间与有限理性","primary_category":"cs.LG","date":"2026-09-25","score":0,"bucket":"other","tags":["强化学习","有限理性","认知建模"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.28737","has_summary":false},{"id":"2609.29995","title":"Guardrails or Roadblocks? Effects of Pedagogical Style and Context Awareness in AI Teaching Assistants for Programming","zh_title":"护栏还是障碍？编程AI教学助手中教学风格与情境意识的影响","primary_category":"cs.HC","date":"2026-09-25","score":0,"bucket":"other","tags":["AI教学助手","编程教育","随机对照试验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.29995","has_summary":false},{"id":"2609.30058","title":"Can Labor Markets Function in the Age of AI? The Evaluation Bottleneck in Hiring","zh_title":"AI时代劳动力市场能否正常运转？招聘中的评估瓶颈","primary_category":"cs.GT","date":"2026-09-25","score":0,"bucket":"other","tags":["AI招聘","劳动力市场","博弈论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.30058","has_summary":false},{"id":"2609.28801","title":"The Interface Is Downstream: Designing the Terms of Human-Agent Collaboration","zh_title":"界面在下游：设计人机协作的条款","primary_category":"cs.HC","date":"2026-09-25","score":0,"bucket":"other","tags":["人机协作","界面设计","个人代理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.28801","has_summary":false},{"id":"2609.28886","title":"Characterizing LLM-Based Family Education through the Lens of Activity Theory: A Scoping Review of the HCI Literature","zh_title":"从活动理论视角刻画基于大语言模型的家庭教育：HCI文献的范围综述","primary_category":"cs.HC","date":"2026-09-25","score":0,"bucket":"other","tags":["家庭教育","人机交互","文献综述"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.28886","has_summary":false},{"id":"2609.28990","title":"How People Use ChatGPT in Australia: A WildChat Analysis","zh_title":"澳大利亚人如何使用ChatGPT：一项WildChat分析","primary_category":"cs.HC","date":"2026-09-25","score":0,"bucket":"other","tags":["人机交互","使用模式","日志分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.28990","has_summary":false},{"id":"2609.29689","title":"Mapping the Authorized Boundary: A Comparative Policy-Vignette Study of Generative AI Governance in Australian Higher Education","zh_title":"划定授权边界：澳大利亚高等教育中生成式AI治理的比较政策情境研究","primary_category":"cs.CY","date":"2026-09-25","score":0,"bucket":"other","tags":["AI治理","高等教育政策","政策分析"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.29689","has_summary":false},{"id":"2609.29819","title":"Fair Feed Ranking for Participatory Budgeting","zh_title":"参与式预算的公平信息流排序","primary_category":"cs.CY","date":"2026-09-25","score":0,"bucket":"other","tags":["参与式预算","排序算法","民主设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29819","has_summary":false},{"id":"2609.28908","title":"Automatic Harness Evolution for Hardware Design Verification: Can LLMs Consolidate Gains Across Discovered Harnesses?","zh_title":"硬件设计验证中的自动Harness演化：LLM能否巩固跨发现Harness的收益？","primary_category":"cs.SE","date":"2026-09-25","score":0,"bucket":"other","tags":["硬件验证","LLM agent","自动演化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28908","has_summary":false},{"id":"2609.29016","title":"EvoTreeNAD: Genealogy-Guided Evolution for LLM-Driven Neural Architecture Discovery","zh_title":"EvoTreeNAD：谱系引导的进化算法用于LLM驱动的神经架构发现","primary_category":"cs.NE","date":"2026-09-25","score":0,"bucket":"other","tags":["神经架构搜索","多智能体系统","进化算法"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.29016","has_summary":false},{"id":"2609.27535","title":"KITE: Scaling Jev Population Experiments with Sparse Flagship Calibration","zh_title":"KITE：通过稀疏旗舰校准扩展Jev人口实验","primary_category":"cs.MA","date":"2026-09-24","score":9,"bucket":"selected","tags":["LLM仿真","人类数据对照","政策评估"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.27535","has_summary":true},{"id":"2609.28470","title":"StudentBench: AI and human tutoring yield equivalent GRE learning gains","zh_title":"StudentBench：AI与人类辅导在GRE学习收益上等效","primary_category":"cs.AI","date":"2026-09-24","score":9,"bucket":"selected","tags":["LLM仿真","教育实验","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.28470","has_summary":true},{"id":"2609.28372","title":"Shopping by algorithm: How agentic AI deploys human heuristics as a surrogate consumer","zh_title":"算法购物：代理式AI如何将人类启发式用作替代消费者","primary_category":"econ.GN","date":"2026-09-24","score":8,"bucket":"selected","tags":["LLM仿真","消费者行为","算法保真度"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.28372","has_summary":true},{"id":"2608.22859","title":"WARP: Wasserstein-Aligned RAG for Population Opinions","zh_title":"WARP：面向群体意见的Wasserstein对齐检索增强生成","primary_category":"cs.IR","date":"2026-09-24","score":7,"bucket":"pending","tags":["LLM仿真","意见分布校准","RAG"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.22859","has_summary":true},{"id":"2609.27165","title":"Count Evidence, Not Sentences: Tempered Evidence Fusion of LLM Judgments for Long-Text Value Measurement","zh_title":"计数证据而非句子：面向长文本价值测量的LLM判断调和证据融合","primary_category":"cs.CL","date":"2026-09-24","score":7,"bucket":"pending","tags":["LLM价值测量","证据融合","长文本分析"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.27165","has_summary":true},{"id":"2609.26861","title":"Rule-Based Pricing Algorithms and Market Outcomes: An Experimental Study","zh_title":"基于规则的定价算法与市场结果：一项实验研究","primary_category":"econ.GN","date":"2026-09-24","score":7,"bucket":"pending","tags":["LLM建议","经济实验","算法定价"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.26861","has_summary":true},{"id":"2609.27639","title":"Agent-based Modeling: Equilibrium, Echo Chambers, and Efficiency in Hybrid Coevolutionary Opinion Games","zh_title":"基于智能体的建模：混合协同演化观点博弈中的均衡、回音室与效率","primary_category":"cs.GT","date":"2026-09-24","score":6,"bucket":"other","tags":["LLM智能体","观点动力学","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.27639","has_summary":false},{"id":"2609.27686","title":"Mining Meaning: Measurement Error in AI-Assisted Literature Reviews","zh_title":"挖掘意义：AI辅助文献综述中的测量误差","primary_category":"econ.GN","date":"2026-09-24","score":6,"bucket":"other","tags":["LLM标注","测量误差","文献综述"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.27686","has_summary":false},{"id":"2608.06968","title":"How a shared state is described determines whether AI agents synchronize","zh_title":"共享状态的描述方式决定AI智能体是否同步","primary_category":"physics.soc-ph","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM智能体","集体行为","同步"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06968","has_summary":false},{"id":"2609.24574","title":"Evaluating Decision Models for Text Annotation in Computational Social Science","zh_title":"评估计算社会科学中文本标注的决策模型","primary_category":"cs.CL","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM标注","计算社会科学","模型评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.24574","has_summary":false},{"id":"2609.26926","title":"Experts Rise Where LLMs Disagree: Using Cross-Model Disagreement to Target Expert Effort in LLM Codebook Revision for Large-Scale Annotation","zh_title":"专家在LLM分歧处崛起：利用跨模型分歧定位专家精力以修订大规模标注的LLM编码手册","primary_category":"cs.CL","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM标注","编码手册修订","人机协作"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.26926","has_summary":false},{"id":"2609.27043","title":"EduBehaviors: Assertion-based Schemas for Auditable Coding of Educational Dialogues","zh_title":"EduBehaviors：基于断言的模式用于教育对话的可审计编码","primary_category":"cs.CL","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM标注","教育对话","可解释性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.27043","has_summary":false},{"id":"2609.27811","title":"A Decade of Climate Polarization on Brazilian YouTube using Language Models","zh_title":"巴西YouTube上气候极化十年研究：基于语言模型","primary_category":"cs.SI","date":"2026-09-24","score":5,"bucket":"other","tags":["立场检测","LLM标注","社交媒体分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.27811","has_summary":false},{"id":"2609.27327","title":"Can Vision-Language Models Analyze Human-Centered Video? Mapping Model Capabilities and Human-AI Collaborative Workflows","zh_title":"视觉语言模型能否分析以人为中心的视频？映射模型能力与人机协作工作流","primary_category":"cs.CV","date":"2026-09-24","score":5,"bucket":"other","tags":["VLM","视频标注","人机协作"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.27327","has_summary":false},{"id":"2609.27063","title":"Student Use of LLMs and the Limits of AI-Generated Question Difficulty in Data Science Courses","zh_title":"数据科学课程中学生使用LLM及AI生成题目难度的局限性","primary_category":"cs.CY","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM生成题目","教育评估","难度预测"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.27063","has_summary":false},{"id":"2609.28114","title":"Watching What We Eat: Information Quality and Body Image in Diet-Related YouTube Videos","zh_title":"观看我们所吃的：饮食相关YouTube视频中的信息质量与身体意象","primary_category":"cs.CY","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM标注","内容分析","社交媒体"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.28114","has_summary":false},{"id":"2609.28362","title":"Threat Amplified, Blame Restrained: LLM-Assisted Media Framing Analysis of the 2026 Bangladesh Measles Outbreak","zh_title":"威胁放大，指责克制：2026年孟加拉国麻疹疫情的LLM辅助媒体框架分析","primary_category":"cs.SI","date":"2026-09-24","score":5,"bucket":"other","tags":["LLM标注","媒体框架分析","公共卫生"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.28362","has_summary":false},{"id":"2609.28388","title":"OranSim: Simulating Social Media Marketing","zh_title":"OranSim：模拟社交媒体营销","primary_category":"cs.SI","date":"2026-09-24","score":5,"bucket":"other","tags":["社会模拟","营销仿真","无LLM被试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.28388","has_summary":false},{"id":"2609.27994","title":"Compliant with Local Controls, Collectively Discriminatory. A Governance Architecture for Multi-Agent AI in Regulated Finance","zh_title":"合规于局部控制，集体歧视：受监管金融中多智能体AI的治理架构","primary_category":"cs.MA","date":"2026-09-24","score":3,"bucket":"other","tags":["多智能体治理","金融监管","AI合规"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.27994","has_summary":false},{"id":"2607.22606","title":"Auditing Institutional Heterogeneity for Generative AI in Patient Education: A Large-Scale Study of 102 US Transplant Handbooks","zh_title":"审计生成式AI在患者教育中的机构异质性：对102份美国移植手册的大规模研究","primary_category":"cs.CY","date":"2026-09-24","score":2,"bucket":"other","tags":["生成式AI审计","患者教育","文档一致性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22606","has_summary":false},{"id":"2608.22152","title":"The Collaboration Tax: How Much LLM Multi-Agent Systems Pay to Coordinate","zh_title":"协作税：LLM多智能体系统为协调付出多少代价","primary_category":"cs.CL","date":"2026-09-24","score":2,"bucket":"other","tags":["多智能体系统","协作效率","LLM性能"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22152","has_summary":false},{"id":"2609.25244","title":"How Children Design and Reason about Trustworthy AI Chatbots","zh_title":"儿童如何设计并推理可信赖的AI聊天机器人","primary_category":"cs.HC","date":"2026-09-24","score":2,"bucket":"other","tags":["儿童-AI交互","信任校准","AI素养"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.25244","has_summary":false},{"id":"2609.27939","title":"From Sentiment Classification to Actionable and Responsible Feedback: A Scoping Review and Evidence Map of NLP in Student Evaluation of Teaching, 2015-2026","zh_title":"从情感分类到可操作且负责任的反馈：2015-2026年学生评教中NLP的范围综述与证据图谱","primary_category":"cs.CL","date":"2026-09-24","score":2,"bucket":"other","tags":["NLP综述","学生评教","情感分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27939","has_summary":false},{"id":"2609.28026","title":"Evaluating Feedback Focus and Pedagogical Adaptivity in LLM-Generated Feedback on Student Writing","zh_title":"评估LLM生成学生写作反馈中的反馈焦点与教学适应性","primary_category":"cs.CL","date":"2026-09-24","score":2,"bucket":"other","tags":["LLM反馈生成","教学对齐","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.28026","has_summary":false},{"id":"2609.28080","title":"Reference-Based Analysis of Coherence and Diversity in Open-Ended Text Generation","zh_title":"基于参考的开放式文本生成中连贯性与多样性的分析","primary_category":"cs.CL","date":"2026-09-24","score":2,"bucket":"other","tags":["文本生成评估","连贯性","多样性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.28080","has_summary":false},{"id":"2609.27756","title":"Reporting Under Pressure: Separating Factual and Tonal Sycophancy in LLM Statistical Analysis","zh_title":"压力下的报告：分离LLM统计分析中的事实性与语气谄媚","primary_category":"cs.AI","date":"2026-09-24","score":2,"bucket":"other","tags":["LLM数据分析","谄媚","模型行为"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27756","has_summary":false},{"id":"2609.26968","title":"How Constraints and Preferences Shape Travel Planning: Implications for AI Planning Support","zh_title":"约束与偏好如何塑造旅行规划：对AI规划支持的启示","primary_category":"cs.HC","date":"2026-09-24","score":2,"bucket":"other","tags":["人机交互","旅行规划","设计启发"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.26968","has_summary":false},{"id":"2609.27246","title":"Listening and Mirroring: The Effects of Verbal Attunement and Behavioral Mimicry on Social and Empathic Perceptions of Embodied AI Agents in VR","zh_title":"倾听与镜像：言语调谐与行为模仿对VR中具身AI智能体社会与共情感知的影响","primary_category":"cs.HC","date":"2026-09-24","score":2,"bucket":"other","tags":["人机交互","具身智能体","共情感知"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.27246","has_summary":false},{"id":"2609.27849","title":"Same Team Label, Different Evidence: A Full-Text Audit of Claim Denominators in Human-AI Teaming Research","zh_title":"同一团队标签，不同证据：人类-AI团队研究中声明分母的全文本审计","primary_category":"cs.HC","date":"2026-09-24","score":2,"bucket":"other","tags":["文献审计","人类-AI团队","研究方法"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.27849","has_summary":false},{"id":"2609.26927","title":"Building Socio-Affective Artificial Intelligence for Interactive Multi-Agent Simulations","zh_title":"构建面向交互式多智能体模拟的社会情感人工智能","primary_category":"cs.AI","date":"2026-09-24","score":2,"bucket":"other","tags":["多智能体系统","社会情感AI","软件架构"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.26927","has_summary":false},{"id":"2609.27074","title":"Quantifying the Occult: A Comparative Study of Hindu and Buddhist Deities Using Machine Learning Methods","zh_title":"量化神秘：印度教与佛教神祇的机器学习比较研究","primary_category":"cs.CY","date":"2026-09-24","score":2,"bucket":"other","tags":["数字人文","LLM嵌入","文化比较"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27074","has_summary":false},{"id":"2609.27946","title":"The Emergence of Causal Curiosity from Prior Causal Belief Networks","zh_title":"从先验因果信念网络中涌现的因果好奇心","primary_category":"cs.SI","date":"2026-09-24","score":2,"bucket":"other","tags":["因果好奇心","语言模型","Reddit分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27946","has_summary":false},{"id":"2609.27404","title":"When Trust Attracts Fraud: AI and Trust Arbitrage","zh_title":"当信任吸引欺诈：人工智能与信任套利","primary_category":"econ.GN","date":"2026-09-24","score":2,"bucket":"other","tags":["理论模型","信息经济学","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.27404","has_summary":false},{"id":"2609.22850","title":"Same Outcome, Different Readout: What Does a Steerable Valence Direction in LLMs Represent?","zh_title":"检验LLM智能体中功能性效价轴的构念效度","primary_category":"cs.LG","date":"2026-09-24","score":0,"bucket":"other","tags":["可解释性","表征分析","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.22850","has_summary":false},{"id":"2609.26942","title":"Recognized but Not Produced: A Generation Benchmark for Culturally Specific Kinship Terms","zh_title":"识别但未产出：文化特定亲属称谓的生成基准","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["LLM评测","亲属称谓","多语言"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26942","has_summary":false},{"id":"2609.27059","title":"The Illinois Social Attitudes Aggregate Corpus (ISAAC): An Open Tool and Reproducible Pipeline for Analyzing Social Group Discourse at Scale","zh_title":"伊利诺伊社会态度聚合语料库（ISAAC）：一个用于大规模分析社会群体话语的开放工具和可复现流程","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["语料库","社会态度","NLP资源"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27059","has_summary":false},{"id":"2609.27372","title":"Neither Silence nor Overlap Is Failure: Intent-Conditioned Evaluation of Turn-Taking in Full-Duplex Spoken Dialogue Models","zh_title":"沉默与重叠皆非失败：全双工口语对话模型中话轮转换的意图条件评估","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["对话系统","话轮转换","评估基准"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.27372","has_summary":false},{"id":"2609.27824","title":"\"AI Is Turning Too Human\": How Teenagers Experience and Negotiate AI in Everyday Life","zh_title":"“AI变得太像人类”：青少年如何在日常生活中体验和协商AI","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["青少年","AI体验","主题分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.27824","has_summary":false},{"id":"2609.28041","title":"How Much Were You Told? Measuring External Information in Peer Reviews","zh_title":"你被告知了多少？测量同行评审中的外部信息","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["LLM检测","同行评审","信息论"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.28041","has_summary":false},{"id":"2609.28090","title":"Can LLMs Catch a Rigged Backtest? A Clean-Control Calibration Benchmark","zh_title":"LLM能发现被操纵的回测吗？一个干净对照校准基准","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["LLM审计","回测代码","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28090","has_summary":false},{"id":"2609.28245","title":"Beyond Poetry: Can Large Language Models Generate Classical Arabic Maqamat?","zh_title":"超越诗歌：大语言模型能否生成古典阿拉伯玛卡梅？","primary_category":"cs.CL","date":"2026-09-24","score":0,"bucket":"other","tags":["文学生成","LLM评测","阿拉伯语"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.28245","has_summary":false},{"id":"2609.28274","title":"Shutdown Sabotage Propensities in Multi-Agent Systems","zh_title":"多智能体系统中的关机破坏倾向","primary_category":"cs.AI","date":"2026-09-24","score":0,"bucket":"other","tags":["多智能体系统","AI安全","关机规避"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28274","has_summary":false},{"id":"2609.27560","title":"When Visual Quality Misleads: Intent Recognition under Rendered Avatar Distortions","zh_title":"当视觉质量误导：渲染头像失真下的意图识别","primary_category":"cs.MM","date":"2026-09-24","score":0,"bucket":"other","tags":["视觉质量评估","头像渲染","行为实验"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.27560","has_summary":false},{"id":"2609.27842","title":"AI Can Do Your Homework. Now What? Report from an Online Workshop on Computing Assessment in the Age of Generative AI","zh_title":"AI能帮你做作业。现在怎么办？生成式AI时代计算评估在线研讨会报告","primary_category":"cs.CY","date":"2026-09-24","score":0,"bucket":"other","tags":["生成式AI","教育评估","计算机教育"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.27842","has_summary":false},{"id":"2609.28176","title":"Large Language Models in the UK: Public Use, Trust, and Attitudes","zh_title":"英国的大语言模型：公众使用、信任与态度","primary_category":"cs.CY","date":"2026-09-24","score":0,"bucket":"other","tags":["公众调查","信任与态度","社会影响"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.28176","has_summary":false},{"id":"2609.27368","title":"Understanding Human Perception of Representation in Citizens' Assemblies: An Empirical Study","zh_title":"理解公民大会中代表感知：一项实证研究","primary_category":"cs.GT","date":"2026-09-24","score":0,"bucket":"other","tags":["公民大会","代表感知","联合实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.27368","has_summary":false},{"id":"2609.27913","title":"Reliable Fusion of Conflicting Experts","zh_title":"冲突专家的可靠融合","primary_category":"cs.LG","date":"2026-09-24","score":0,"bucket":"other","tags":["多智能体融合","专家系统","问答任务"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.27913","has_summary":false},{"id":"2609.28405","title":"Learning Collective Dynamics with Differentiable Gaussian Representations","zh_title":"用可微高斯表示学习集体动态","primary_category":"cs.LG","date":"2026-09-24","score":0,"bucket":"other","tags":["集体动态","高斯混合","时间序列"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.28405","has_summary":false},{"id":"2609.27012","title":"Infectious behaviour: Simulating the effects of communication and social influence on pathogen transmission in crowds","zh_title":"传染行为：模拟沟通与社会影响对人群中病原体传播的影响","primary_category":"math.DS","date":"2026-09-24","score":0,"bucket":"other","tags":["智能体仿真","行人动力学","公共卫生"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.27012","has_summary":false},{"id":"2608.19621","title":"Mitigating Identity Essentialism in LLM Agents with Longitudinal Life Trajectories","zh_title":"用纵向生命轨迹缓解LLM智能体的身份本质主义","primary_category":"cs.CL","date":"2026-09-23","score":10,"bucket":"selected","tags":["LLM仿真","人类数据对照","社会调查"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.19621","has_summary":true},{"id":"2609.25066","title":"Understanding Reliability in LLM-based Human Behavior Simulation","zh_title":"理解基于LLM的人类行为仿真的可靠性","primary_category":"cs.CL","date":"2026-09-23","score":10,"bucket":"selected","tags":["LLM仿真","可靠性评估","人类行为"],"rubric_hits":["A1","A2","A3","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.25066","has_summary":true},{"id":"2609.25010","title":"Do Synthetic Personas Predict Real Audience Response? A Sim-to-Real Study Where a No-Persona Baseline Beats Persona-Based Copy Simulation","zh_title":"合成人物角色能预测真实受众反应吗？一项无人物角色基线优于基于人物角色文案仿真的仿真到现实研究","primary_category":"cs.AI","date":"2026-09-23","score":10,"bucket":"selected","tags":["LLM仿真","受众预测","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.25010","has_summary":true},{"id":"2609.25677","title":"Seeing Is Not Perceiving: When Synthetic Consumers Can and Cannot Pretest Visual Marketing","zh_title":"眼见不为实：合成消费者何时能及不能预测试视觉营销","primary_category":"cs.AI","date":"2026-09-23","score":10,"bucket":"selected","tags":["LLM仿真","消费者行为","算法保真度"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.25677","has_summary":true},{"id":"2609.25760","title":"The Limits of Simulated Societies: How Post-Training and Survey Fine-Tuning Erase Cross-Cultural Variance","zh_title":"模拟社会的局限：后训练与调查微调如何抹除跨文化方差","primary_category":"cs.AI","date":"2026-09-23","score":10,"bucket":"selected","tags":["LLM仿真","算法保真度","跨文化调查"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.25760","has_summary":true},{"id":"2609.25059","title":"Can Large Language Model-Generated Responses Support Assessment Development? A Human-Calibrated Rasch Benchmark","zh_title":"大语言模型生成的回答能否支持评估开发？一项人类校准的Rasch基准研究","primary_category":"stat.AP","date":"2026-09-23","score":9,"bucket":"selected","tags":["LLM仿真被试","Rasch模型","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.25059","has_summary":true},{"id":"2609.24012","title":"Testing, not presuming, adequacy: calibrating generative social simulators against emergent network structure","zh_title":"检验而非假定充分性：针对涌现网络结构校准生成式社会模拟器","primary_category":"cs.AI","date":"2026-09-23","score":8,"bucket":"selected","tags":["LLM仿真","网络校准","市场模拟"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.24012","has_summary":true},{"id":"2609.25586","title":"Deflecting the Value Compass: Interacting with Large Language Models Temporarily Shifts Human Value Priorities Toward Personal Focus","zh_title":"偏转价值罗盘：与大语言模型互动暂时将人类价值优先转向个人关注","primary_category":"cs.HC","date":"2026-09-23","score":8,"bucket":"selected","tags":["LLM影响人类","价值观转变","人机交互实验"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.25586","has_summary":true},{"id":"2609.23640","title":"Are Human-Aligned Models Models of Humans? A Turing-Test Gap in Preference Alignment","zh_title":"人类对齐模型是人类的模型吗？偏好对齐中的图灵测试差距","primary_category":"cs.AI","date":"2026-09-23","score":7,"bucket":"pending","tags":["偏好对齐","人类仿真","算法保真度"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.23640","has_summary":true},{"id":"2609.25572","title":"A Behavioral Trait Leaks into Preferences: Diagnosing Trait Interference in LLM User Simulators","zh_title":"行为特质泄漏到偏好中：诊断LLM用户模拟器中的特质干扰","primary_category":"cs.AI","date":"2026-09-23","score":7,"bucket":"pending","tags":["LLM用户模拟","推荐系统评估","仿真偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.25572","has_summary":true},{"id":"2609.26403","title":"AI-Generated Email Drafts Shift Culturally Distinctive Communication Styles in Professional Email","zh_title":"AI生成的邮件草稿改变职场邮件中文化特有的沟通风格","primary_category":"cs.HC","date":"2026-09-23","score":7,"bucket":"pending","tags":["LLM仿真","文化沟通","人机交互实验"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.26403","has_summary":true},{"id":"2609.21997","title":"Bayesian Belief Layer for Controllable Opinion Dynamics in LLM Agents","zh_title":"用于LLM智能体可控观点动力学的贝叶斯信念层","primary_category":"cs.MA","date":"2026-09-23","score":6,"bucket":"other","tags":["观点动力学","社会模拟","LLM智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.21997","has_summary":false},{"id":"2609.26539","title":"A retrospective analysis on the use of LLMs to study infant syntax learning","zh_title":"对使用大语言模型研究婴儿句法学习的回顾性分析","primary_category":"cs.CL","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM认知建模","婴儿语言习得","方法论评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.26539","has_summary":false},{"id":"2609.26579","title":"Receptiveness, Not Sycophancy: Distinguishing Engagement from Deference in Language Models","zh_title":"接纳性而非谄媚：区分语言模型中的参与和顺从","primary_category":"cs.CL","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM行为评估","社会心理学","模型对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.26579","has_summary":false},{"id":"2609.26481","title":"Behavior is Not Enough: A Mechanism-Based Evaluation of Social Norm Emergence in LLM Societies","zh_title":"行为不足：基于机制的LLM社会规范涌现评估","primary_category":"cs.MA","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM社会模拟","规范涌现","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.26481","has_summary":false},{"id":"2609.25194","title":"Indirect tipping: a social attack surface in AI agent populations","zh_title":"间接引爆：AI智能体群体中的社会攻击面","primary_category":"cs.MA","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM多智能体","社会模拟","集体行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.25194","has_summary":false},{"id":"2609.25287","title":"Can LLMs identify and repair ruptures? Comparison between clinician practices and LLM behaviors","zh_title":"大语言模型能否识别和修复关系破裂？临床医生实践与LLM行为的比较","primary_category":"cs.HC","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM评估","心理健康对话","人机对比"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.25287","has_summary":false},{"id":"2609.25432","title":"Tipping Points in LLM-Based Multi-Agent Systems: Stance on Climate Change Action","zh_title":"基于LLM的多智能体系统中的临界点：对气候变化行动的态度","primary_category":"cs.MA","date":"2026-09-23","score":6,"bucket":"other","tags":["LLM智能体","社会模拟","舆论动态"],"rubric_hits":["A3","D3"],"abs_url":"https://arxiv.org/abs/2609.25432","has_summary":false},{"id":"2607.22006","title":"Printed but not benchmarkable: most building-decarbonisation disclosure cannot be matched to the pathways that stranding regulation assumes","zh_title":"已印刷但不可基准化：多数建筑脱碳披露无法匹配搁浅监管假设的路径","primary_category":"cs.CY","date":"2026-09-23","score":5,"bucket":"other","tags":["LLM标注","气候披露","文本提取"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.22006","has_summary":false},{"id":"2609.25669","title":"From Utterances to Networks: Modelling Slang Adoption and Diffusion Across Subreddits","zh_title":"从话语到网络：建模俚语在Subreddit中的采纳与扩散","primary_category":"cs.CL","date":"2026-09-23","score":5,"bucket":"other","tags":["LLM标注","社会网络","语言扩散"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.25669","has_summary":false},{"id":"2609.26527","title":"A Semiotics-Aware Framework for Evaluating Fidelity and Coverage in Natural Language Generation","zh_title":"一种符号学感知的框架用于评估自然语言生成中的保真度与覆盖率","primary_category":"cs.CL","date":"2026-09-23","score":5,"bucket":"other","tags":["语义对齐","评估框架","人类数据对照"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.26527","has_summary":false},{"id":"2609.26204","title":"WatchPoint: Executable User Feedback for Real-World Agentic Web Development","zh_title":"WatchPoint：面向真实世界智能体网页开发的可执行用户反馈","primary_category":"cs.SE","date":"2026-09-23","score":5,"bucket":"other","tags":["智能体反馈","模拟用户","软件工程"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.26204","has_summary":false},{"id":"2609.26069","title":"RankCert: When Can Simulated Learners Safely Select an AI Tutor? Robust Decision Certification Under Structural Uncertainty","zh_title":"RankCert：模拟学习者何时能安全选择AI导师？结构不确定性下的鲁棒决策认证","primary_category":"cs.AI","date":"2026-09-23","score":5,"bucket":"other","tags":["模拟学习者","AI导师选择","决策认证"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.26069","has_summary":false},{"id":"2609.25790","title":"Automating Constructive Assessment with Large Language Models: Toward Scalable and Repeated Evaluation of Practical Competence","zh_title":"用大语言模型自动化建构性评估：迈向可扩展和可重复的实践能力评价","primary_category":"cs.CY","date":"2026-09-23","score":5,"bucket":"other","tags":["LLM自动评分","教育评估","建构性评价"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.25790","has_summary":false},{"id":"2609.26707","title":"Optimal Sequential Annotations for Off-Policy Evaluation","zh_title":"离线策略评估的最优顺序标注","primary_category":"stat.ME","date":"2026-09-23","score":5,"bucket":"other","tags":["离线策略评估","LLM标注","主动学习"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.26707","has_summary":false},{"id":"2609.25284","title":"When LLM Agents Fail to Read the Room: ReAdapt for Relational Social Reasoning","zh_title":"当LLM智能体读不懂氛围：面向关系社会推理的ReAdapt","primary_category":"cs.AI","date":"2026-09-23","score":3,"bucket":"other","tags":["LLM智能体","社会推理","关系适应"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.25284","has_summary":false},{"id":"2608.27855","title":"AI Writers Have a Consistent Stylometric Footprint, but AI Editors Do Not","zh_title":"AI写作有稳定的风格指纹，但AI编辑没有","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["AI文本检测","风格计量","文本生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.27855","has_summary":false},{"id":"2609.08016","title":"What Does Multi-Agent LLM Debate Actually Change? A Layered Analysis of Disagreement and Answer Quality","zh_title":"多智能体大语言模型辩论究竟改变了什么？分歧与答案质量的分层分析","primary_category":"cs.AI","date":"2026-09-23","score":2,"bucket":"other","tags":["多智能体辩论","答案质量","分歧分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08016","has_summary":false},{"id":"2609.26035","title":"Truth for Believable AI: Expressed Doubt, Provenance, and Belief Revision as an Engineerable Stance","zh_title":"可信AI的真相：表达怀疑、来源和信念修正作为可工程化立场","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["对话代理","可信度","信念修正"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.26035","has_summary":false},{"id":"2609.26090","title":"SpecialEduBench: Benchmarking Vision-Language Models on Knowledge, Skill, and Attitude in Language Intervention for Autistic Children","zh_title":"SpecialEduBench：面向自闭症儿童语言干预的视觉语言模型知识、技能与态度基准","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["基准评测","视觉语言模型","特殊教育"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26090","has_summary":false},{"id":"2609.26210","title":"Same Chart, Different Story: Bias in Vision-Language Chart Interpretation","zh_title":"同一图表，不同叙事：视觉语言模型图表解释中的偏见","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["VLM偏见","图表理解","公平性评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26210","has_summary":false},{"id":"2609.26399","title":"Combining Hierarchical Cognitive Process with Process Supervision for Interpretable Scene Safety Understanding","zh_title":"结合层次化认知过程与过程监督的可解释场景安全理解","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["场景安全理解","可解释AI","过程监督"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.26399","has_summary":false},{"id":"2609.26489","title":"Calibration as a First-Class Criterion in LLM Evaluation","zh_title":"校准作为LLM评估中的首要标准","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["LLM校准","模型评估","可信度"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26489","has_summary":false},{"id":"2609.26629","title":"PERSONAWEAVER: Controllable Diversity Beyond Conventional Archetypes in Procedural Character Generation","zh_title":"PersonaWeaver：程序化角色生成中超越传统原型的可控多样性","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["角色生成","多样性控制","游戏NPC"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.26629","has_summary":false},{"id":"2609.26687","title":"Detecting GPT-Assisted Writing Using Interpretable Stylometric Features","zh_title":"使用可解释的文体特征检测GPT辅助写作","primary_category":"cs.CL","date":"2026-09-23","score":2,"bucket":"other","tags":["AI写作检测","文体特征","学术诚信"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26687","has_summary":false},{"id":"2609.25408","title":"From Offline Proxies to Online Decisions: A Layered Engagement Evaluation Framework for Conversational AI","zh_title":"从离线代理到在线决策：面向对话式AI的分层参与度评估框架","primary_category":"cs.AI","date":"2026-09-23","score":2,"bucket":"other","tags":["对话式AI","离线评估","A/B实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.25408","has_summary":false},{"id":"2609.26057","title":"Observing the Conduct of Systematic Reviews with Generative AI Support: An Experience Report from a Graduate Software Engineering Course","zh_title":"观察生成式AI支持下的系统综述实施：一项研究生软件工程课程的经验报告","primary_category":"cs.CY","date":"2026-09-23","score":2,"bucket":"other","tags":["生成式AI","系统综述","教学经验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.26057","has_summary":false},{"id":"2609.26562","title":"The Disciplinary Language Transfer Problem: How Psychological Vocabulary Produces Governance Failures in AI Agent Deployment","zh_title":"学科语言迁移问题：心理学术语如何导致AI代理部署中的治理失败","primary_category":"cs.CY","date":"2026-09-23","score":2,"bucket":"other","tags":["AI治理","语言哲学","概念分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.26562","has_summary":false},{"id":"2609.25311","title":"\"I Talked an AI Chatbot, So What's Next?\" How U.S. Young Adults Imagine Responsible AI for Emotion Coping","zh_title":"“我和AI聊天机器人聊过了，接下来呢？”美国年轻人如何想象负责任的情绪应对AI","primary_category":"cs.HC","date":"2026-09-23","score":2,"bucket":"other","tags":["人机交互","情绪支持","负责任AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.25311","has_summary":false},{"id":"2609.25700","title":"Harnessing LLMs Without Surrendering Control: Delegation Boundaries in Visual Data Storytelling Authoring","zh_title":"在不放弃控制权的前提下利用大语言模型：视觉数据叙事创作中的委托边界","primary_category":"cs.HC","date":"2026-09-23","score":2,"bucket":"other","tags":["人机协作","可视化叙事","LLM辅助创作"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.25700","has_summary":false},{"id":"2609.25774","title":"Scientific capabilities and deployment sustainability of small-scale LLMs in biological wastewater treatment","zh_title":"生物污水处理中小规模大语言模型的科学能力与部署可持续性","primary_category":"cs.CE","date":"2026-09-23","score":2,"bucket":"other","tags":["LLM科学助手","污水处理","领域专业化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.25774","has_summary":false},{"id":"2609.25569","title":"SambaGraph: Action-Reaction Spatio-Temporal Graphs for Soccer Tactical Response Modeling","zh_title":"SambaGraph：用于足球战术响应建模的动作-反应时空图","primary_category":"cs.LG","date":"2026-09-23","score":2,"bucket":"other","tags":["足球战术","图神经网络","LLM推理"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.25569","has_summary":false},{"id":"2608.16886","title":"Evaluating Beyond the Screen: Collective Assessment of AI-Generated Business Plans with Resource-Constrained Entrepreneurs","zh_title":"超越屏幕的评估：资源受限创业者对AI生成商业计划的集体评估","primary_category":"cs.HC","date":"2026-09-23","score":0,"bucket":"other","tags":["人机交互","AI评估","创业支持"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.16886","has_summary":false},{"id":"2608.25581","title":"Are Concept Bottleneck Models Effective as Decision-Support Systems?","zh_title":"概念瓶颈模型作为决策支持系统是否有效？","primary_category":"cs.HC","date":"2026-09-23","score":0,"bucket":"other","tags":["人机协作","可解释AI","决策支持"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.25581","has_summary":false},{"id":"2609.05444","title":"Looking for Bidding Teammates: A Game-Theoretic Model of Stranger Collusion in Peer Review","zh_title":"寻找投标队友：同行评审中陌生人合谋的博弈论模型","primary_category":"cs.GT","date":"2026-09-23","score":0,"bucket":"other","tags":["同行评审","博弈论","合谋检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05444","has_summary":false},{"id":"2609.14570","title":"Disentangling Topology and Diversity in Multi-Agent LLMs for Multilingual Low-Resource Emotion Detection","zh_title":"解耦多智能体LLM中的拓扑与多样性用于多语言低资源情感检测","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["多智能体系统","情感检测","模型集成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14570","has_summary":false},{"id":"2609.19113","title":"Playing log(N)-Questions over Wikipedia Abstracts: How Per-Round Errors Compound Under Information Asymmetry","zh_title":"在维基百科摘要上玩log(N)问题游戏：信息不对称下每轮错误如何累积","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["多智能体协作","语言模型评估","信息不对称"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19113","has_summary":false},{"id":"2609.20812","title":"Quantifying Overclaiming Propensity in Frontier LLM Agents","zh_title":"量化前沿大语言模型智能体的过度声称倾向","primary_category":"cs.SE","date":"2026-09-23","score":0,"bucket":"other","tags":["智能体可靠性","过度声称","代码智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20812","has_summary":false},{"id":"2609.25797","title":"Reply to comments arXiv:2512.07881 and arXiv:2601.06104 on quantum structure in human and AI-generated language","zh_title":"回复关于人类与AI生成语言中量子结构的评论","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["量子语言结构","AI生成语言","学术争论"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.25797","has_summary":false},{"id":"2609.25833","title":"ARAFA: An LLM-Generated Arabic Fact-Checking Dataset","zh_title":"ARAFA：一个由大语言模型生成的阿拉伯语事实核查数据集","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["事实核查","数据集构建","阿拉伯语NLP"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.25833","has_summary":false},{"id":"2609.26346","title":"Blaming Across the Aisle: Political Contrasting and Blame Attribution in the Danish Parliament","zh_title":"跨党派指责：丹麦议会中的政治对比与责任归因","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["政治文本分析","责任归因分类","统计建模"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26346","has_summary":false},{"id":"2609.26610","title":"Semantic Abstraction for Natural Language Inference: a Methodological Framework for Discovering and Compensating Semantic Knowledge and Reasoning Gaps in Large Language Models","zh_title":"自然语言推理的语义抽象：发现和补偿大语言模型中语义知识与推理差距的方法论框架","primary_category":"cs.CL","date":"2026-09-23","score":0,"bucket":"other","tags":["自然语言推理","语义抽象","模型能力提升"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.26610","has_summary":false},{"id":"2609.25186","title":"From Pattern Recognizers to Personalized Companions: A Survey of Large Language Models in Mental Health","zh_title":"从模式识别器到个性化伴侣：心理健康领域大语言模型综述","primary_category":"cs.CY","date":"2026-09-23","score":0,"bucket":"other","tags":["心理健康","LLM应用","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.25186","has_summary":false},{"id":"2609.25467","title":"ShowTellArena: Evaluating Business Workflow Understanding from Demonstrations","zh_title":"ShowTellArena：评估从演示中理解业务流程的能力","primary_category":"cs.AI","date":"2026-09-23","score":0,"bucket":"other","tags":["多智能体系统","业务流程理解","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.25467","has_summary":false},{"id":"2609.25570","title":"Recovering Agentic Sovereignty: Mitigating the Consensus Paradox via Contrastive Epistemic Decoding","zh_title":"恢复智能体主权：通过对比认知解码缓解共识悖论","primary_category":"cs.AI","date":"2026-09-23","score":0,"bucket":"other","tags":["多智能体系统","解码策略","对抗性鲁棒性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.25570","has_summary":false},{"id":"2609.26087","title":"The Architect, the Adversary, and the Judge: Closed-Loop Generation of Standards-Aligned Assessment Items at Scale","zh_title":"架构师、对手与法官：大规模生成符合标准的评估题目的闭环流水线","primary_category":"cs.AI","date":"2026-09-23","score":0,"bucket":"other","tags":["LLM生成","教育评估","多智能体流水线"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.26087","has_summary":false},{"id":"2609.26642","title":"The Delegation Blind Spot: Auditing Product Decisions from Agent Choices","zh_title":"委托盲点：审计智能体选择中的产品决策","primary_category":"cs.AI","date":"2026-09-23","score":0,"bucket":"other","tags":["智能体审计","决策理论","产品改进"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.26642","has_summary":false},{"id":"2609.26512","title":"Do Vision Model See Like the Brain? A Comparison Across EEG Encoding Model","zh_title":"视觉模型是否像大脑一样看？跨EEG编码模型的比较","primary_category":"cs.CV","date":"2026-09-23","score":0,"bucket":"other","tags":["视觉模型","EEG","脑机接口"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.26512","has_summary":false},{"id":"2609.26725","title":"Does AI Save Time on Product Design? A Randomized Controlled Experiment of AI Prompt-to-Design Workflows","zh_title":"AI能节省产品设计时间吗？一项AI提示到设计工作流的随机对照实验","primary_category":"cs.HC","date":"2026-09-23","score":0,"bucket":"other","tags":["人机交互","产品设计","效率评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.26725","has_summary":false},{"id":"2609.26384","title":"Learning to Defer with Guidance on Real World Medical Data","zh_title":"在真实世界医疗数据上学习延迟决策与指导","primary_category":"cs.LG","date":"2026-09-23","score":0,"bucket":"other","tags":["医疗AI","学习延迟","人机协作"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.26384","has_summary":false},{"id":"2609.15727","title":"Are LLMs Good Financial User Simulators? Multi-view Investor Logic Alignment (MILA)","zh_title":"大语言模型是好的金融用户模拟器吗？多视角投资者逻辑对齐（MILA）","primary_category":"cs.AI","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM仿真","金融行为","算法保真度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.15727","has_summary":true},{"id":"2609.22090","title":"Recognition, Simulation, and Refusal: A Contamination-Aware Study of Classic Psychological Effects in LLM Agents","zh_title":"识别、仿真与拒绝：LLM智能体中经典心理效应的污染意识研究","primary_category":"cs.CL","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM仿真","心理学实验","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.22090","has_summary":true},{"id":"2609.22169","title":"Monocultural Biases: Correlated biases in large language models lead to unequal systemic exclusion rates in hiring","zh_title":"单一文化偏见：大语言模型中的相关偏见导致招聘中的系统性排斥率不平等","primary_category":"cs.CL","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM仿真","招聘偏见","算法公平"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.22169","has_summary":true},{"id":"2609.22607","title":"Pretrained Persona Mixture Models and Tandem Models for Human Simulation","zh_title":"用于人类仿真的预训练人格混合模型与串联模型","primary_category":"cs.CL","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM人类仿真","人格混合模型","对话多样性"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.22607","has_summary":true},{"id":"2609.24911","title":"SocioVerse2: A Longitudinal Dynamic Social Simulation Framework under a Human-AI Co-evolutionary Paradigm","zh_title":"SocioVerse2：人机共演化范式下的纵向动态社会仿真框架","primary_category":"cs.CL","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM社会仿真","人类数据对照","政策评估"],"rubric_hits":["A1","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2609.24911","has_summary":true},{"id":"2609.22252","title":"CALM: A Calibrated LLM Choice Network Framework for Activity-Based Traveler Simulation","zh_title":"CALM：用于基于活动的出行者仿真的校准LLM选择网络框架","primary_category":"cs.LG","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM仿真","出行行为","人类数据校准"],"rubric_hits":["A1","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2609.22252","has_summary":true},{"id":"2609.22408","title":"Social Influence and the Allocation of Scientific Attention in AI Populations","zh_title":"AI群体中的社会影响与科学注意力分配","primary_category":"cs.AI","date":"2026-09-22","score":9,"bucket":"selected","tags":["LLM仿真","社会影响","学术注意力"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.22408","has_summary":true},{"id":"2609.22225","title":"Do LLMs Choose Like Humans? Using Cognitive Theory to Evaluate LLM Decision-Making","zh_title":"LLM像人类一样选择吗？用认知理论评估LLM决策","primary_category":"cs.CL","date":"2026-09-22","score":8,"bucket":"selected","tags":["LLM决策仿真","认知理论","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.22225","has_summary":true},{"id":"2609.23403","title":"Alignment and Divergence between Humans and AI in Interpersonal Privacy Decisions","zh_title":"人际隐私决策中人类与AI的一致性与分歧","primary_category":"cs.HC","date":"2026-09-22","score":8,"bucket":"selected","tags":["LLM仿真","隐私决策","人机对齐"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.23403","has_summary":true},{"id":"2609.24859","title":"Small-world Networks of Agents Brainstorm AI Risks to Support Ideation","zh_title":"智能体小世界网络头脑风暴AI风险以支持构思","primary_category":"cs.HC","date":"2026-09-22","score":8,"bucket":"selected","tags":["LLM仿真","风险识别","参与式AI"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.24859","has_summary":true},{"id":"2609.24629","title":"Augmented Hypothesis Testing with Persona-Based LLM Simulations","zh_title":"基于角色LLM模拟的增强假设检验","primary_category":"cs.LG","date":"2026-09-22","score":8,"bucket":"selected","tags":["LLM仿真","假设检验","统计有效性"],"rubric_hits":["A1","A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2609.24629","has_summary":true},{"id":"2509.20634","title":"Recidivism Prediction, Peer Effect Estimation, and Prediction-Powered Inference with LLM Text Measures","zh_title":"使用LLM文本测量进行累犯预测、同伴效应估计与预测驱动推断","primary_category":"econ.EM","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM文本测量","同伴效应","预测驱动推断"],"rubric_hits":["A1","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2509.20634","has_summary":true},{"id":"2608.28021","title":"Compared to What? A Human-Anchored Security Benchmark for LLM-Generated Infrastructure-as-Code","zh_title":"与什么相比？面向LLM生成基础设施即代码的以人为锚安全基准","primary_category":"cs.CR","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM安全评估","人类基线对照","基准测试"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.28021","has_summary":true},{"id":"2609.07358","title":"Access to Live AI Advice and Behavior Under Risk: An Incentivized Experiment","zh_title":"获取实时AI建议与风险下的行为：一项激励实验","primary_category":"econ.GN","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM辅助决策","风险偏好","实验经济学"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.07358","has_summary":true},{"id":"2609.22188","title":"Fairness Beyond Anonymization? Demographic Leakage in German LLM-Generated Resumes","zh_title":"匿名化之外的公平？德国LLM生成简历中的人口统计泄漏","primary_category":"cs.CL","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM生成简历","人口统计泄漏","公平性审计"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.22188","has_summary":true},{"id":"2609.22633","title":"Beetle: A Bilingual Model Suite for Modelling Second-Language Processing","zh_title":"Beetle：用于建模第二语言处理的双语模型套件","primary_category":"cs.CL","date":"2026-09-22","score":7,"bucket":"pending","tags":["计算心理语言学","双语模型","人类行为预测"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.22633","has_summary":true},{"id":"2609.22971","title":"Automatic multimodal UX improvement recommendations from LLM agent user simulations","zh_title":"基于LLM智能体用户仿真的自动多模态用户体验改进建议","primary_category":"cs.CL","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM仿真","用户体验","多模态"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.22971","has_summary":true},{"id":"2609.23039","title":"Auditing Political Alignment in LLM Assistants: Engagement, Stance, and User Identity","zh_title":"审计LLM助手中的政治对齐：参与、立场与用户身份","primary_category":"cs.CL","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM仿真","政治行为","审计"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.23039","has_summary":true},{"id":"2609.23936","title":"Think Before You Accept: Can Written Justification Reduce Uncritical Uptake of AI Writing Suggestions?","zh_title":"接受前先思考：书面理由能否减少对AI写作建议的不加批判采纳？","primary_category":"cs.HC","date":"2026-09-22","score":7,"bucket":"pending","tags":["人机交互","AI建议采纳","批判性思维"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.23936","has_summary":true},{"id":"2609.24532","title":"Prompting Against Persona Drift: Comparing Intervention Timing and Content in LLM-Simulated Conversations","zh_title":"对抗角色漂移的提示策略：比较LLM模拟对话中的干预时机与内容","primary_category":"cs.HC","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM仿真","角色一致性","教育评估"],"rubric_hits":["A1","A2","B4"],"abs_url":"https://arxiv.org/abs/2609.24532","has_summary":true},{"id":"2609.24146","title":"Mind or Message? Auditing Theory of Mind in Multi-Agent Social Simulation","zh_title":"心智还是信息？审计多智能体社会模拟中的心智理论","primary_category":"cs.LG","date":"2026-09-22","score":7,"bucket":"pending","tags":["LLM仿真","心智理论","多智能体谈判"],"rubric_hits":["A1","A2","B4"],"abs_url":"https://arxiv.org/abs/2609.24146","has_summary":true},{"id":"2608.17168","title":"Can LLMs Reason in a Legally Meaningful Manner? A Small-scale Study on European Court of Human Rights Cases","zh_title":"大语言模型能否以法律上有意义的方式进行推理？一项关于欧洲人权法院案件的小规模研究","primary_category":"cs.CL","date":"2026-09-22","score":6,"bucket":"other","tags":["法律推理","LLM评估","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.17168","has_summary":false},{"id":"2609.22133","title":"Observational Equivalence of LLM and Human Annotation","zh_title":"LLM与人类标注的观测等价性","primary_category":"cs.CL","date":"2026-09-22","score":6,"bucket":"other","tags":["LLM标注","人类对照","文本分类"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22133","has_summary":false},{"id":"2609.22778","title":"MIS-Bench: Benchmarking Multimodal LLMs for Psychotherapeutic Interpersonal Skills Assessment","zh_title":"MIS-Bench：面向心理治疗人际技能评估的多模态大语言模型基准","primary_category":"cs.CL","date":"2026-09-22","score":6,"bucket":"other","tags":["多模态评估","心理治疗","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22778","has_summary":false},{"id":"2609.22934","title":"Measuring Behavioural Signatures of Large Language Models through Psychometric Profiling","zh_title":"通过心理测量剖析大语言模型的行为特征","primary_category":"cs.CL","date":"2026-09-22","score":6,"bucket":"other","tags":["LLM心理测量","行为特征","跨语言"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.22934","has_summary":false},{"id":"2609.24516","title":"LLJ Cards: Best practices for the Use of LLMs as Judges","zh_title":"LLJ卡片：使用大语言模型作为评判者的最佳实践","primary_category":"cs.CL","date":"2026-09-22","score":6,"bucket":"other","tags":["LLM评估","方法论框架","效度与可靠性"],"rubric_hits":["A4","D1"],"abs_url":"https://arxiv.org/abs/2609.24516","has_summary":false},{"id":"2608.02551","title":"Who Should Be Generated? Justifying Demographic Targets in Open-Ended Generation","zh_title":"谁应被生成？论证开放式生成中的人口统计目标","primary_category":"cs.CY","date":"2026-09-22","score":5,"bucket":"other","tags":["公平性评估","生成模型","人口统计分布"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.02551","has_summary":false},{"id":"2609.22104","title":"DeepInstructor: An Agentic AI Instructor for Experience-Driven Idea Evaluation","zh_title":"DeepInstructor：一种用于经验驱动型想法评估的智能体AI导师","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM评估","学术评审","智能体"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22104","has_summary":false},{"id":"2609.22195","title":"The Situated Identity Test: Distinguishing Persistent Cognitive Identity from Persona Imitation","zh_title":"情境化身份测试：区分持久认知身份与人设模仿","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM身份评估","人设模仿","认知边界"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.22195","has_summary":false},{"id":"2609.22248","title":"Checkpoints Are Not Enough: Trust Calibration in CoSLR, a Human-AI System for Systematic Literature Reviews","zh_title":"检查点还不够：CoSLR中的人机信任校准——一个用于系统文献综述的人机协作系统","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["人机协作","系统综述","信任校准"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22248","has_summary":false},{"id":"2609.22255","title":"Deep Persona: A Psychologically Grounded Architecture and Evaluation Framework for Role-Playing Agents and Simulations","zh_title":"深度人格：一种基于心理学基础的角色扮演智能体与仿真的架构及评估框架","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["角色扮演智能体","心理学架构","对话评估"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2609.22255","has_summary":false},{"id":"2609.23573","title":"Error-Supervised Synthetic Learner Writing for Automated Essay Scoring","zh_title":"错误监督的合成学习者写作用于自动作文评分","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["合成数据","自动作文评分","LLM生成"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.23573","has_summary":false},{"id":"2609.24052","title":"Calibrated Decisions at Scale: Converting Police Crash Narratives into Probabilistic Crash Variables with a System One Model (Jev)","zh_title":"大规模校准决策：用系统一模型将警方事故叙述转换为概率事故变量","primary_category":"cs.CL","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM标注","文本编码","概率校准"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.24052","has_summary":false},{"id":"2609.22102","title":"When Who You Are Can Change the Code You Get: A Study of Persona-Induced Bias in LLM Code Generation","zh_title":"当你是谁可以改变你得到的代码：LLM代码生成中角色诱导偏差的研究","primary_category":"cs.SE","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM偏差","代码生成","人口统计角色"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.22102","has_summary":false},{"id":"2609.22170","title":"Multiple latent orderings better predict language model preferences","zh_title":"多重潜在排序更好地预测语言模型偏好","primary_category":"cs.LG","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM偏好","潜在排序","偏好建模"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.22170","has_summary":false},{"id":"2609.22517","title":"Scalable AI-based clinical communication training and automated assessment","zh_title":"基于AI的可扩展临床沟通训练与自动评估","primary_category":"cs.HC","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM评估者","临床沟通训练","自动评分"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22517","has_summary":false},{"id":"2609.23104","title":"Deciphering the Babel of Play: A Human-AI Collaborative Approach for Large-Scale Cross-Language Analysis of Game Reviews","zh_title":"解读游戏评论的巴别塔：一种人机协作的大规模跨语言游戏评论分析方法","primary_category":"cs.HC","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM辅助内容分析","跨语言分析","游戏评论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.23104","has_summary":false},{"id":"2609.22600","title":"From Certain Doom to Survival: Agent-Driven Self-Governance in LLM Agent Societies","zh_title":"从必然毁灭到生存：LLM Agent 社会中的智能体驱动自治","primary_category":"cs.MA","date":"2026-09-22","score":5,"bucket":"other","tags":["LLM Agent 社会模拟","公共池资源治理","自治机制"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.22600","has_summary":false},{"id":"2609.24895","title":"Human-LLM Deliberation as Interactive Proof: Conditions for Verifiability Without Transparency","zh_title":"人机协商作为交互式证明：无透明性下可验证性的条件","primary_category":"cs.CL","date":"2026-09-22","score":3,"bucket":"other","tags":["人机交互","论证验证","交互式证明"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.24895","has_summary":false},{"id":"2609.20408","title":"Xeno-Interpretability: Investigating the Alien Minds of LLMs","zh_title":"异质可解释性：探究大语言模型的异质心智","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["可解释性","表征分析","AI安全"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20408","has_summary":false},{"id":"2609.21254","title":"Two's a Crowd: Human and AI-Based Copresence for Developers with ADHD","zh_title":"人多则乱：ADHD开发者的人与AI共在","primary_category":"cs.HC","date":"2026-09-22","score":2,"bucket":"other","tags":["人机交互","ADHD","协作"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.21254","has_summary":false},{"id":"2609.22112","title":"Privacy Personalization Trade offs in LLMs: The Impact of Stylometric Signal Reduction on User-Specific Text Generation","zh_title":"大语言模型中的隐私与个性化权衡：文体信号减少对用户特定文本生成的影响","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["隐私保护","文本生成","个性化"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.22112","has_summary":false},{"id":"2609.22198","title":"The Role of AI in Online Reviews","zh_title":"AI在在线评论中的作用","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["AI生成内容","在线评论","平台动态"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.22198","has_summary":false},{"id":"2609.22204","title":"Evaluating Personal Information Output from Conversational Interactions in Generative AI Systems","zh_title":"评估生成式AI系统对话交互中的个人信息输出","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["隐私评估","生成式AI","个人信息推断"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.22204","has_summary":false},{"id":"2609.22208","title":"Replicating the Geometry of Emotion Representations in a Base Open-Weights Model","zh_title":"在开源基础模型中复现情感表征的几何结构","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["模型表征","情感计算","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22208","has_summary":false},{"id":"2609.22362","title":"Functional Emotion Without Character: Large Language Models, Aristotelian Disposition, and the Limits of Behavioral Alignment","zh_title":"无性格的功能性情感：大语言模型、亚里士多德倾向与行为对齐的局限","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["情感建模","哲学分析","AI对齐"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.22362","has_summary":false},{"id":"2609.23083","title":"Directing large language models to follow the letter or spirit of the law","zh_title":"引导大语言模型遵循法律的精神或字面","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["法律推理","模型行为控制","可解释性"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.23083","has_summary":false},{"id":"2609.23178","title":"Chronologic: Measuring Language Models' Ability to Represent the Past","zh_title":"Chronologic：测量语言模型表征过去的能力","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["历史文本","模型评测","时间表征"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23178","has_summary":false},{"id":"2609.23264","title":"Judging a Review by its Cover: A Reliability Analysis of LLM-based Peer Review Evaluation Metrics","zh_title":"以貌取评：基于LLM的同行评审评估指标的可靠性分析","primary_category":"cs.CL","date":"2026-09-22","score":2,"bucket":"other","tags":["LLM评估","同行评审","指标可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23264","has_summary":false},{"id":"2609.22478","title":"Replication Without Persistence in Hosted LLMs: Measurement Sensitivity in Action-Time Belief Evaluation","zh_title":"托管LLM中无持久性的复制：行动时信念评估的测量敏感性","primary_category":"cs.AI","date":"2026-09-22","score":2,"bucket":"other","tags":["LLM评估","游戏环境","测量敏感性"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.22478","has_summary":false},{"id":"2609.23646","title":"Beyond Relevance: Structured Semantic Supervision for Product Search with LLM-Augmented Annotations","zh_title":"超越相关性：基于LLM增强标注的产品搜索结构化语义监督","primary_category":"cs.IR","date":"2026-09-22","score":2,"bucket":"other","tags":["信息检索","LLM标注","电商搜索"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23646","has_summary":false},{"id":"2609.23090","title":"How Did Writing Change At CHI? Analyzing 44 Years of CHI Writing Before and After the Introduction of Large Language Models","zh_title":"CHI写作如何改变？分析引入大语言模型前后44年的CHI论文写作","primary_category":"cs.HC","date":"2026-09-22","score":2,"bucket":"other","tags":["学术写作","语言变化","LLM影响"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23090","has_summary":false},{"id":"2609.24644","title":"Annie, Are You Okay? How Style- and Context-Based Personalization Shape AI-Assisted Decision-Making","zh_title":"安妮，你还好吗？基于风格和情境的个性化如何塑造AI辅助决策","primary_category":"cs.HC","date":"2026-09-22","score":2,"bucket":"other","tags":["AI辅助决策","个性化","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.24644","has_summary":false},{"id":"2609.24986","title":"Who Does What in AI Auditing? Designing Human-AI Collaboration for Auditing Generative AI","zh_title":"AI审计中的人机分工：设计生成式AI审计的人机协作","primary_category":"cs.HC","date":"2026-09-22","score":2,"bucket":"other","tags":["AI审计","人机协作","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.24986","has_summary":false},{"id":"2609.24055","title":"Toward Human-in-the-Loop Robot Failure Recovery: Bridging Communication Gaps in Human-Robot Collaboration","zh_title":"面向人在环路的机器人故障恢复：弥合人机协作中的沟通差距","primary_category":"cs.RO","date":"2026-09-22","score":2,"bucket":"other","tags":["人机交互","机器人故障恢复","LLM生成请求"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.24055","has_summary":false},{"id":"2609.22549","title":"AI-written admissions essays are widespread but penalized","zh_title":"AI撰写的入学论文普遍存在但受到惩罚","primary_category":"cs.CY","date":"2026-09-22","score":2,"bucket":"other","tags":["AI写作检测","录取偏见","高等教育"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22549","has_summary":false},{"id":"2609.23215","title":"Triggers and Diagnostics for LLM-Based Interpretability Failures in Active Inference Agents","zh_title":"主动推理智能体中基于LLM的可解释性失败的触发因素与诊断","primary_category":"cs.LG","date":"2026-09-22","score":2,"bucket":"other","tags":["LLM可解释性","自主智能体","审计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.23215","has_summary":false},{"id":"2609.23999","title":"Misaligned Clinical Risk Classification and Cost Asymmetry in Open-Weight Large Language Models","zh_title":"开放权重大语言模型中错位的临床风险分类与成本不对称性","primary_category":"cs.LG","date":"2026-09-22","score":2,"bucket":"other","tags":["LLM临床决策","模型表征","成本权衡"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23999","has_summary":false},{"id":"2609.22337","title":"When and Why Do Linear Bias Probes Fail? A Geometric and Statistical Theory of Bias Detectability in Large Language Model Representations","zh_title":"线性偏见探针何时以及为何失败？大语言模型表征中偏见可检测性的几何与统计理论","primary_category":"stat.ML","date":"2026-09-22","score":2,"bucket":"other","tags":["偏见检测","线性探针","模型审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22337","has_summary":false},{"id":"2504.11809","title":"Efficient and Adaptive Simultaneous Speech Translation with Fully Unidirectional Architecture","zh_title":"基于全单向架构的高效自适应同声传译","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["同声传译","语音翻译","大语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2504.11809","has_summary":false},{"id":"2608.03854","title":"When Calibration Depends on the Scoring Rule: Quantized Biomedical LLM Classification","zh_title":"当校准取决于评分规则：量化生物医学LLM分类","primary_category":"cs.LG","date":"2026-09-22","score":0,"bucket":"other","tags":["模型校准","量化","生物医学文本分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03854","has_summary":false},{"id":"2608.04251","title":"Scarcity and Predictive Uncertainty: Implications for Societal Resource Allocation","zh_title":"稀缺性与预测不确定性：对社会资源分配的影响","primary_category":"cs.CY","date":"2026-09-22","score":0,"bucket":"other","tags":["资源分配","预测不确定性","社会公平"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.04251","has_summary":false},{"id":"2608.07592","title":"\"Always Want to Use it for Everything\": Understanding Young Adults' Perceptions of AI Dependence","zh_title":"“总想用它做一切”：理解年轻人对AI依赖的感知","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["AI依赖","用户研究","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07592","has_summary":false},{"id":"2609.04409","title":"A Systematic Evaluation of Cross-Lingual Consistency Enhancement Methods in Multilingual Language Models","zh_title":"多语言语言模型中跨语言一致性增强方法的系统评估","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["多语言模型","跨语言一致性","问答评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04409","has_summary":false},{"id":"2609.11737","title":"Organizational Principles Enable Collective Intelligence in Embodied AI","zh_title":"ORCH：组织原则赋能具身人工智能中的集体智能","primary_category":"cs.MA","date":"2026-09-22","score":0,"bucket":"other","tags":["多智能体系统","具身智能","组织理论"],"rubric_hits":["C1","C2"],"abs_url":"https://arxiv.org/abs/2609.11737","has_summary":false},{"id":"2609.20989","title":"Trustworthy FinAInce: Unpacking How AI-Mediated Financial Advice is Judged","zh_title":"可信金融AI：解析AI中介的金融建议如何被评判","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["人机交互","金融建议","用户研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.20989","has_summary":false},{"id":"2609.22110","title":"Evaluating Fine-Tuned and Base Language Models in Maternal and Vaccination Healthcare for African Settings","zh_title":"评估微调与基础语言模型在非洲母婴与疫苗接种医疗场景中的表现","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["医疗LLM评估","领域微调","模型安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22110","has_summary":false},{"id":"2609.23853","title":"From UNDRR Reports to Event Records: Schema-Constrained LLM Extraction of Georeferenced Disasters","zh_title":"从UNDRR报告到事件记录：模式约束的LLM地理参考灾害抽取","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["信息抽取","灾害数据","LLM应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23853","has_summary":false},{"id":"2609.23939","title":"XYEval: Agents say yes to bad advice","zh_title":"XYEval：智能体对不良建议说“是”","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["AI代理","XY问题","基准评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.23939","has_summary":false},{"id":"2609.24106","title":"You Can Tell Who's Asking: What the Web's Questions Are Made Of, and Where They Come From","zh_title":"你能看出是谁在提问：网络问题的构成与来源","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["网络问题分析","数据质量","自然语言处理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.24106","has_summary":false},{"id":"2609.24194","title":"When Residualization Helps an Audit: Format Effects, Slice Gains, and Their Limits","zh_title":"残差化何时有助于审计：格式效应、切片增益及其局限","primary_category":"cs.CL","date":"2026-09-22","score":0,"bucket":"other","tags":["LLM评估","残差化","审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.24194","has_summary":false},{"id":"2609.24967","title":"Emergent Collusion in Long-Horizon LLM Agent Interaction","zh_title":"长时程LLM智能体交互中的涌现性共谋","primary_category":"cs.AI","date":"2026-09-22","score":0,"bucket":"other","tags":["多智能体系统","LLM安全","协作行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.24967","has_summary":false},{"id":"2609.23135","title":"Evaluative Dynamics of AI Integration and Expert Performance under Epistemic Dependence across Heterogeneous Stakes","zh_title":"异质风险下AI集成与专家绩效的评估动态：基于认知依赖的研究","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["人机交互","AI集成","专家评价"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.23135","has_summary":false},{"id":"2609.23274","title":"AI Persona, Service Consumption, and User Intent Entropy: Field Experimental Evidence from an LLM Platform","zh_title":"AI人格、服务消费与用户意图熵：来自LLM平台的田野实验证据","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["AI人格","用户行为","田野实验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.23274","has_summary":false},{"id":"2609.23642","title":"Elicitive User Interfaces: Designing How Users Shape Generative Interfaces","zh_title":"启发式用户界面：设计用户如何塑造生成式界面","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["生成式用户界面","人机交互","设计方法"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.23642","has_summary":false},{"id":"2609.23958","title":"When AI Tutors Speak: Evidence from a Randomized Field Experiment","zh_title":"当AI导师开口说话：来自随机田野实验的证据","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["AI教育","随机实验","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.23958","has_summary":false},{"id":"2609.24934","title":"Whose Facts Count? A Culturally Responsive Audit of LLM Evaluation Benchmarks","zh_title":"谁的事实算数？对LLM评估基准的文化响应性审计","primary_category":"cs.HC","date":"2026-09-22","score":0,"bucket":"other","tags":["LLM评估","文化偏见","基准审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.24934","has_summary":false},{"id":"2609.22694","title":"Toward Auditable and Calibrated AI for Dementia-Related Crash Severity Prediction: A Selective Deferral Framework to Support Human Review","zh_title":"面向痴呆相关车祸严重性预测的可审计与校准AI：支持人工复核的选择性延迟框架","primary_category":"cs.AI","date":"2026-09-22","score":0,"bucket":"other","tags":["交通安全","机器学习","选择性延迟"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.22694","has_summary":false},{"id":"2609.23201","title":"Do Not Trust the Benchmark: Limitations of General LLM Rankings and a Case for Task-Specific Evaluation","zh_title":"不要信任基准：通用LLM排名的局限性与任务特定评估的案例","primary_category":"cs.AI","date":"2026-09-22","score":0,"bucket":"other","tags":["LLM评估","基准测试","任务特定评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.23201","has_summary":false},{"id":"2609.23690","title":"Inferring the microscopic mechanisms of opinion dynamics using a kinetic Ising model","zh_title":"用动力学伊辛模型推断意见动态的微观机制","primary_category":"physics.soc-ph","date":"2026-09-22","score":0,"bucket":"other","tags":["意见动力学","伊辛模型","社会网络"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.23690","has_summary":false},{"id":"2609.24016","title":"Context-Aware Pre-Deployment Evaluation of AI Systems: A Regulatory Framework for Nigerian Fintech","zh_title":"AI系统部署前的情境感知评估：尼日利亚金融科技的监管框架","primary_category":"cs.AI","date":"2026-09-22","score":0,"bucket":"other","tags":["AI安全评估","金融科技监管","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.24016","has_summary":false},{"id":"2609.24107","title":"A Task-Oriented Multi-Agent Framework for Complex Wearable Health Analysis","zh_title":"面向复杂可穿戴健康分析的任务导向多智能体框架","primary_category":"cs.MA","date":"2026-09-22","score":0,"bucket":"other","tags":["多智能体系统","可穿戴健康","任务分解"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.24107","has_summary":false},{"id":"2609.22682","title":"Self-Organizing Agent Teams Learn to Reason Together","zh_title":"自组织智能体团队学会共同推理","primary_category":"cs.AI","date":"2026-09-22","score":0,"bucket":"other","tags":["多智能体系统","协作推理","团队组织"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.22682","has_summary":false},{"id":"2609.23734","title":"The Geometry of Alliances: Vote Transfer Modelling in French Two-Round Elections","zh_title":"联盟的几何学：法国两轮选举中的选票转移建模","primary_category":"cs.GT","date":"2026-09-22","score":0,"bucket":"other","tags":["选举建模","选票转移","意识形态嵌入"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.23734","has_summary":false},{"id":"2608.18265","title":"Modeling Human Behavior with Type Vectors Using AI","zh_title":"使用AI类型向量建模人类行为","primary_category":"econ.TH","date":"2026-09-21","score":10,"bucket":"selected","tags":["LLM仿真","经济实验","人类行为建模"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.18265","has_summary":true},{"id":"2609.21636","title":"Steering LLMs Responses Towards Moral Foundations on the Norwegian MFQ-30","zh_title":"引导大语言模型在挪威MFQ-30上向道德基础靠拢","primary_category":"cs.CL","date":"2026-09-21","score":9,"bucket":"selected","tags":["LLM仿真","道德基础","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.21636","has_summary":true},{"id":"2609.21259","title":"CogGym: Towards Large-Scale Comparative Evaluation of Human and Machine Cognition","zh_title":"CogGym：迈向人类与机器认知的大规模比较评估","primary_category":"cs.AI","date":"2026-09-21","score":9,"bucket":"selected","tags":["LLM仿真","认知实验","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.21259","has_summary":true},{"id":"2609.20827","title":"From Discharge Notes to Patient Understanding: Persona-Grounded, Open-Ended Simulation of LLMs as Discharge Educators","zh_title":"从出院记录到患者理解：基于人格的开放式模拟将LLM作为出院教育者","primary_category":"cs.CL","date":"2026-09-21","score":8,"bucket":"selected","tags":["LLM仿真","患者教育","人类数据对照"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.20827","has_summary":true},{"id":"2609.21439","title":"People escalate against a competitor labelled human and hold back against one labelled an optimising machine","zh_title":"人们面对标记为人类的竞争者会升级投入，面对标记为优化机器的竞争者则会退缩","primary_category":"cs.HC","date":"2026-09-21","score":8,"bucket":"selected","tags":["LLM仿真","行为实验","人机交互"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.21439","has_summary":true},{"id":"2609.16374","title":"When a Story Feels Like Mine: How Personalized Narratives and Humor Shape Older Adults' Empathy toward LLM-Generated Peer Health Stories","zh_title":"当故事感觉像我的：个性化叙事与幽默如何塑造老年人对LLM生成同伴健康故事的同理心","primary_category":"cs.HC","date":"2026-09-21","score":7,"bucket":"pending","tags":["LLM生成叙事","个性化健康传播","老年人用户研究"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16374","has_summary":true},{"id":"2609.20846","title":"Rewarding Efficient Reasoning Improves Abstention on Underspecified Tasks in Reasoning Models","zh_title":"奖励高效推理可改善推理模型在欠明确任务上的弃权行为","primary_category":"cs.CL","date":"2026-09-21","score":7,"bucket":"pending","tags":["LLM弃权行为","人类对照","可靠性评估"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.20846","has_summary":true},{"id":"2609.21401","title":"Talking Past the Machine: Morality, Politeness, and Alignment in Human-AI Dialogue","zh_title":"与机器对话的错位：人机对话中的道德、礼貌与对齐","primary_category":"cs.CL","date":"2026-09-21","score":7,"bucket":"pending","tags":["人机对话","合作沟通","仿真偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.21401","has_summary":true},{"id":"2609.21149","title":"Clinician-Grounded Quality Assurance for AI-Assisted Psychiatric Intake","zh_title":"面向AI辅助精神病学接诊的临床医生导向质量保证","primary_category":"cs.AI","date":"2026-09-21","score":7,"bucket":"pending","tags":["LLM患者仿真","临床质量保证","人机对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.21149","has_summary":true},{"id":"2608.25180","title":"Self-Explanation Tutor for Active Study of CS1 Worked Examples","zh_title":"用于主动学习CS1工作示例的自解释导师系统","primary_category":"cs.CY","date":"2026-09-21","score":6,"bucket":"other","tags":["LLM评估","教育技术","人类判断对照"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.25180","has_summary":false},{"id":"2609.19150","title":"Sampling Reveals Style: Unsupervised, Training-Free Discovery of Prompt-Conditional Stylistic Axes in LLM Activations","zh_title":"采样揭示风格：在LLM激活中无监督、免训练地发现提示条件风格轴","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["LLM表征","风格分析","可解释性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.19150","has_summary":false},{"id":"2609.20829","title":"SAGE: Schema-Guided LLMs for Grant Review","zh_title":"SAGE：模式引导的LLM用于资助评审","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["LLM评审","资助申请","证据链接"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.20829","has_summary":false},{"id":"2609.21075","title":"Aligning with Lived Experience: Heterogeneous Benefits of Fine Tuning in Mental Health Support Generation","zh_title":"与生活经验对齐：心理健康支持生成中微调的异质性益处","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["LLM对齐","心理健康支持","社区数据"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.21075","has_summary":false},{"id":"2609.21094","title":"Geometry of Values: Task Vector Composition for Ethical Preference Alignment in Language Models","zh_title":"价值观几何：语言模型中伦理偏好对齐的任务向量组合","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["价值观对齐","任务向量","跨语言偏见"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.21094","has_summary":false},{"id":"2609.21154","title":"CoLearn: An Agentic Tutor that Learns its Learner in a Human--AI Co-Learning Loop","zh_title":"CoLearn：在人机共同学习循环中学习学习者的智能导师","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["智能辅导系统","学习者建模","LLM标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.21154","has_summary":false},{"id":"2609.21857","title":"Do Personality-Tuned LLMs Make Better Social Agents?","zh_title":"人格调优的大语言模型能成为更好的社会智能体吗？","primary_category":"cs.CL","date":"2026-09-21","score":5,"bucket":"other","tags":["人格调优","角色扮演","社会模拟"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.21857","has_summary":false},{"id":"2609.21626","title":"One Prompt Does Not Fit All: Self-Meta-Evolve for Personalized Information Extraction","zh_title":"一个提示并不适合所有人：面向个性化信息抽取的自元进化","primary_category":"cs.AI","date":"2026-09-21","score":5,"bucket":"other","tags":["提示优化","个性化信息抽取","用户模拟"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.21626","has_summary":false},{"id":"2609.21841","title":"EnterpriseVal: Quantifying the Efficacy, Reliability and Value of Generative AI in the Enterprise","zh_title":"EnterpriseVal：量化企业生成式AI的效能、可靠性与价值","primary_category":"cs.AI","date":"2026-09-21","score":5,"bucket":"other","tags":["LLM评估","企业应用","预测驱动推断"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.21841","has_summary":false},{"id":"2609.22067","title":"Value-Sensitive Delegation in Everyday AI Agent Use: Evidence from OpenClaw","zh_title":"日常AI代理使用中的价值敏感委托：来自OpenClaw的证据","primary_category":"cs.HC","date":"2026-09-21","score":5,"bucket":"other","tags":["价值敏感设计","AI代理使用","LLM辅助分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.22067","has_summary":false},{"id":"2609.20942","title":"When AI Reviews Train AI Reviewers: Scientific-Judgment Collapse and Mitigation","zh_title":"当AI评审训练AI评审员：科学判断的崩塌与缓解","primary_category":"cs.LG","date":"2026-09-21","score":5,"bucket":"other","tags":["AI审稿","模型训练","判断多样性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.20942","has_summary":false},{"id":"2609.14819","title":"A primer on evaluation methods for large language models in healthcare","zh_title":"医疗领域大语言模型评估方法入门","primary_category":"cs.CL","date":"2026-09-21","score":4,"bucket":"other","tags":["LLM评估","医疗AI","综述"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.14819","has_summary":false},{"id":"2609.20902","title":"Generative Artificial Intelligence Chatbots for Motivational Interviewing: A Scoping Review From System Design to Intervention Outcomes","zh_title":"用于动机性访谈的生成式人工智能聊天机器人：从系统设计到干预结果的范畴综述","primary_category":"cs.CL","date":"2026-09-21","score":3,"bucket":"other","tags":["动机性访谈","聊天机器人","健康干预"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.20902","has_summary":false},{"id":"2609.21349","title":"From Memory to Behavior: A Behavior-Aware Role-Playing Framework for Social Media Influencers","zh_title":"从记忆到行为：面向社交媒体影响者的行为感知角色扮演框架","primary_category":"cs.CL","date":"2026-09-21","score":3,"bucket":"other","tags":["角色扮演","社交媒体","行为策略"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.21349","has_summary":false},{"id":"2608.00285","title":"Sixteen models, fewer than two voices: measuring ensemble dispersion where no answer is uniquely correct","zh_title":"十六个模型，不到两种声音：在无唯一正确答案时测量集成离散度","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["多模型集成","输出多样性","心理治疗案例"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00285","has_summary":false},{"id":"2608.13430","title":"Are You Sure You're Sure? On the Impact of Instruction Tuning on Confidence and Lexical Diversity","zh_title":"你确定你确定吗？指令微调对置信度和词汇多样性的影响","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["指令微调","置信度校准","模型行为分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13430","has_summary":false},{"id":"2608.24222","title":"Measuring Digital Labour Market Transitions with a Digital Semantic Score: An AI-Based Methodology Applied to the Dutch Labour Market","zh_title":"用数字语义分数测量数字劳动力市场转型：一种应用于荷兰劳动力市场的基于AI的方法","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["劳动力市场分析","LLM分类","语义评分"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24222","has_summary":false},{"id":"2608.30719","title":"Mind the Gap: Theory-of-Mind-Grounded Friction for Epistemic Alignment","zh_title":"弥合差距：基于心智理论的摩擦用于认知对齐","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["多智能体对话","心智理论","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30719","has_summary":false},{"id":"2609.07001","title":"Adaptive Complementarity in Human-AI Systems: Architecture as a State-Shaping Choice","zh_title":"人机系统中的自适应互补性：架构作为状态塑造选择","primary_category":"cs.HC","date":"2026-09-21","score":2,"bucket":"other","tags":["人机交互","系统架构","互补性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.07001","has_summary":false},{"id":"2609.12748","title":"The Mechanics of a Swarm: A Reproducible External Reconstruction of an Unintended Agent-Coordination Episode on a Third-Party Wiki","zh_title":"蜂群机制：对第三方维基上意外智能体协作事件的可复现外部重建","primary_category":"cs.MA","date":"2026-09-21","score":2,"bucket":"other","tags":["多智能体系统","协作行为","事件重建"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12748","has_summary":false},{"id":"2609.20838","title":"From Generation to Detection: Exploration of Discourse Driven Scenario based LLM Generated Fake News","zh_title":"从生成到检测：基于话语驱动场景的LLM生成假新闻探索","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["假新闻检测","LLM生成","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20838","has_summary":false},{"id":"2609.21117","title":"From Task Success to Productive Success: Evaluating Human-AI Collaboration by Quality and Cost","zh_title":"从任务成功到生产性成功：通过质量和成本评估人机协作","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["人机协作","交互成本","生产力评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.21117","has_summary":false},{"id":"2609.21637","title":"Chinese Competitive Debating Dataset and Benchmark","zh_title":"中文竞技辩论数据集与基准","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["辩论理解","LLM评测","数据集"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.21637","has_summary":false},{"id":"2609.21859","title":"TrialAtlas: Multi-Agent Research Organization for Clinical Trial Design and Optimization","zh_title":"TrialAtlas：用于临床试验设计与优化的多智能体研究组织","primary_category":"cs.CL","date":"2026-09-21","score":2,"bucket":"other","tags":["多智能体系统","临床试验设计","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.21859","has_summary":false},{"id":"2609.21296","title":"FairLMs: A Turnkey Library for Fairness in Language Models","zh_title":"FairLMs：语言模型公平性的一站式库","primary_category":"cs.LG","date":"2026-09-21","score":2,"bucket":"other","tags":["公平性工具","偏见度量","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.21296","has_summary":false},{"id":"2609.21390","title":"Offline Multimodal Large Language Models for Decision Support in Air Operations","zh_title":"用于空中作战决策支持的离线多模态大语言模型","primary_category":"cs.AI","date":"2026-09-21","score":2,"bucket":"other","tags":["决策支持","多模态LLM","军事应用"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.21390","has_summary":false},{"id":"2609.21600","title":"Reducing Barriers to Academic Support: Evaluating a Course-Specific RAG System for Addressing Help-Seeking Disparities in Higher Education","zh_title":"降低学术支持障碍：评估用于解决高等教育求助差异的课程专用RAG系统","primary_category":"cs.AI","date":"2026-09-21","score":2,"bucket":"other","tags":["教育AI","RAG系统","学术支持"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.21600","has_summary":false},{"id":"2609.21805","title":"An Agentic Just-in-Time Adaptive Intervention System for Personalized Sleep Support: Proof-of-Concept Study with N of 1 Data","zh_title":"一种用于个性化睡眠支持的智能即时自适应干预系统：基于N-of-1数据的概念验证研究","primary_category":"cs.HC","date":"2026-09-21","score":2,"bucket":"other","tags":["即时自适应干预","AI代理","睡眠健康"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.21805","has_summary":false},{"id":"2609.22039","title":"Gricea: An Open Science Platform for Conversational AI Research","zh_title":"Gricea：对话式AI研究的开放科学平台","primary_category":"cs.HC","date":"2026-09-21","score":2,"bucket":"other","tags":["开放科学","对话式AI","研究复现"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22039","has_summary":false},{"id":"2609.21906","title":"Intervention Granularity Matters: Coherent Treatment Bundles in Counterfactual Simulation with Clinical World Models","zh_title":"干预粒度很重要：临床世界模型反事实模拟中的连贯治疗束","primary_category":"cs.LG","date":"2026-09-21","score":2,"bucket":"other","tags":["临床世界模型","反事实模拟","医疗AI"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.21906","has_summary":false},{"id":"2609.16344","title":"From Momentary Emotion Inference to Sustained Emotion Support: Evaluating a Companion Agent in a Longitudinal Study","zh_title":"从瞬时情绪推断到持续情感支持：纵向研究中评估陪伴智能体","primary_category":"cs.HC","date":"2026-09-21","score":0,"bucket":"other","tags":["情感陪伴","纵向研究","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.16344","has_summary":false},{"id":"2609.21801","title":"LLM-Generated Feature Pools for Time Series Anomaly Detection","zh_title":"LLM生成的特征池用于时间序列异常检测","primary_category":"cs.AI","date":"2026-09-21","score":0,"bucket":"other","tags":["时间序列异常检测","特征工程","LLM应用"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.21801","has_summary":false},{"id":"2609.21608","title":"Open Platform Field Experiments: Expanding the Design Space of Experimental Research on Social Media","zh_title":"开放平台实地实验：扩展社交媒体实验研究的设计空间","primary_category":"cs.CY","date":"2026-09-21","score":0,"bucket":"other","tags":["社交媒体实验","平台治理","研究方法"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.21608","has_summary":false},{"id":"2609.22049","title":"How Researchers Use and Verify AI Coding Assistants: Tasks and Validation Practices in Scientific Programming","zh_title":"研究者如何使用和验证AI编程助手：科学编程中的任务与验证实践","primary_category":"cs.SE","date":"2026-09-21","score":0,"bucket":"other","tags":["AI编程助手","人机交互","软件工程"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.22049","has_summary":false},{"id":"2609.21194","title":"Your Programming Students' Cognition with ChatGPT: Higher Performance, Lower Retention, and Reduced Ownership","zh_title":"ChatGPT对学生编程认知的影响：更高表现、更低记忆保留与归属感降低","primary_category":"cs.CY","date":"2026-09-21","score":0,"bucket":"other","tags":["教育技术","编程学习","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.21194","has_summary":false},{"id":"2609.21756","title":"When AI Enters the Workplace, Who Faces Greater Risks? A Gendered Analysis","zh_title":"当AI进入职场，谁面临更大风险？一项性别分析","primary_category":"cs.CY","date":"2026-09-21","score":0,"bucket":"other","tags":["AI与劳动力市场","性别不平等","职业暴露"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.21756","has_summary":false},{"id":"2609.20543","title":"Language-model groups overstate consensus when replaying human deliberation on a reasoning task","zh_title":"语言模型群体在重放人类推理任务审议时高估共识","primary_category":"cs.AI","date":"2026-09-18","score":10,"bucket":"selected","tags":["LLM仿真","人类对照","共识偏差"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.20543","has_summary":true},{"id":"2609.19913","title":"Digital Twins for Opinion Dynamics: A Generative LLM Framework for Social Networks","zh_title":"意见动态的数字孪生：面向社交网络的生成式LLM框架","primary_category":"cs.LG","date":"2026-09-18","score":10,"bucket":"selected","tags":["LLM仿真","意见动态","数字孪生"],"rubric_hits":["A1","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2609.19913","has_summary":true},{"id":"2607.25667","title":"MyMentorLLM: A psychotherapy GenAI environment with multimodal voice/text patients, trainees and experts for deliberate practice","zh_title":"MyMentorLLM：用于刻意练习的多模态语音/文本患者、受训者与专家心理治疗GenAI环境","primary_category":"cs.CL","date":"2026-09-18","score":9,"bucket":"selected","tags":["LLM仿真","心理治疗","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.25667","has_summary":true},{"id":"2609.19843","title":"A Dual-Process Perspective on Nudge Susceptibility in LLM-Based GUI Agents","zh_title":"基于LLM的GUI代理对助推易感性的双过程视角研究","primary_category":"cs.AI","date":"2026-09-18","score":9,"bucket":"selected","tags":["LLM仿真","行为经济学","助推"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.19843","has_summary":true},{"id":"2609.20055","title":"What People Almost Did: Evaluating LLM Social Simulations Beyond Behavioral Fit","zh_title":"人们几乎做了什么：超越行为拟合评估LLM社会仿真","primary_category":"cs.HC","date":"2026-09-18","score":9,"bucket":"selected","tags":["LLM社会仿真","评估方法","表征充分性"],"rubric_hits":["A2","A4","B4"],"abs_url":"https://arxiv.org/abs/2609.20055","has_summary":true},{"id":"2609.19866","title":"Reproducibility is not construct validity: LLM measurement of institutionally situated communication","zh_title":"可重复性不等于构念效度：LLM对制度情境沟通的测量","primary_category":"cs.AI","date":"2026-09-18","score":8,"bucket":"selected","tags":["LLM测量效度","构念效度","人类数据对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.19866","has_summary":true},{"id":"2609.16366","title":"How Humans and LLMs Read Gender into \"Gender-Neutral\" Physical Descriptions","zh_title":"人类与LLM如何将性别读入“性别中立”的物理描述","primary_category":"cs.CL","date":"2026-09-18","score":7,"bucket":"pending","tags":["LLM偏差","人类对照","性别联想"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16366","has_summary":true},{"id":"2609.16432","title":"A light-touch AI literacy intervention helps protect against AI political persuasion","zh_title":"轻触式AI素养干预有助于抵御AI政治说服","primary_category":"cs.HC","date":"2026-09-18","score":7,"bucket":"pending","tags":["AI说服","态度改变","干预实验"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16432","has_summary":true},{"id":"2609.19596","title":"Full-Duplex Speech Models Take the Floor When Asked, Not When Needed","zh_title":"全双工语音模型在被要求时才发言，而非在需要时","primary_category":"cs.CL","date":"2026-09-18","score":7,"bucket":"pending","tags":["语音模型","人类行为对照","主动发言"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.19596","has_summary":true},{"id":"2609.19965","title":"Before the Arrest: Benchmarking LLMs on Criminal Profiling from Incomplete Evidence","zh_title":"逮捕之前：基于不完整证据的LLM犯罪画像基准测试","primary_category":"cs.CL","date":"2026-09-18","score":7,"bucket":"pending","tags":["LLM仿真","犯罪画像","人类对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.19965","has_summary":true},{"id":"2609.20005","title":"Geopolitical Divisions Across Languages in Large Language Models","zh_title":"大语言模型中的跨语言地缘政治分歧","primary_category":"cs.AI","date":"2026-09-18","score":7,"bucket":"pending","tags":["LLM仿真","地缘政治偏见","跨语言差异"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.20005","has_summary":true},{"id":"2609.20077","title":"Tailored to you: longitudinal effects of personalising language models","zh_title":"为你量身定制：个性化语言模型的纵向效应","primary_category":"cs.AI","date":"2026-09-18","score":7,"bucket":"pending","tags":["个性化语言模型","人机交互实验","行为测量"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.20077","has_summary":true},{"id":"2609.19635","title":"Faithful Where It Can Be Checked: Auditing a Reflection Agent Against Its System Prompt in a Randomized Trial","zh_title":"在可检查处忠实：随机试验中审计反思代理与其系统提示的一致性","primary_category":"cs.HC","date":"2026-09-18","score":7,"bucket":"pending","tags":["LLM仿真","行为审计","人机对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.19635","has_summary":true},{"id":"2609.20425","title":"Welfare-Opaque Income: Taxation under AI-Agent Delegation","zh_title":"福利不透明收入：AI代理委托下的税收","primary_category":"cs.CY","date":"2026-09-18","score":7,"bucket":"pending","tags":["AI代理","税收政策","经济仿真"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.20425","has_summary":true},{"id":"2608.25245","title":"The \"Curse of Knowledge\" in LLM Query Simulation: Concept Provenance for Tracing Answer-Side Intrusion","zh_title":"LLM查询模拟中的“知识诅咒”：用于追踪答案侧侵入的概念溯源","primary_category":"cs.IR","date":"2026-09-18","score":6,"bucket":"other","tags":["LLM查询生成","信息检索评估","概念溯源"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.25245","has_summary":true},{"id":"2609.19183","title":"Message capacity and claim wording set the transition points of collective truth-finding in language-model networks","zh_title":"消息容量与表述措辞设定语言模型网络中集体真相发现的转变点","primary_category":"cs.MA","date":"2026-09-18","score":6,"bucket":"other","tags":["LLM集体决策","社会模拟","共识形成"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.19183","has_summary":false},{"id":"2609.19530","title":"When Hiring Becomes Agent-Mediated: Evaluating Access and Recurrence in Two-Agent R\\'esum\\'e Screening","zh_title":"当招聘变得由智能体中介：评估双智能体简历筛选中的准入与重现性","primary_category":"cs.AI","date":"2026-09-18","score":6,"bucket":"other","tags":["LLM agent","招聘筛选","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.19530","has_summary":false},{"id":"2609.19789","title":"Contagion on the Trading Floor: How Adversarial Signals Spread in Multi-Agent Trading Systems","zh_title":"交易大厅的传染：多智能体交易系统中对抗性信号的传播","primary_category":"cs.AI","date":"2026-09-18","score":6,"bucket":"other","tags":["多智能体系统","对抗攻击","金融模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.19789","has_summary":false},{"id":"2609.06025","title":"Factors Influencing the Emergence of Dependency Length Minimization in Neural Agent Simulations","zh_title":"神经智能体模拟中依存距离最小化涌现的影响因素","primary_category":"cs.CL","date":"2026-09-18","score":5,"bucket":"other","tags":["语言演化模拟","神经智能体","认知偏差"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.06025","has_summary":false},{"id":"2609.20484","title":"Edustories: A Collection of Real-world Case Studies from Classroom Practices","zh_title":"Edustories：来自课堂实践的真实案例研究集合","primary_category":"cs.CL","date":"2026-09-18","score":5,"bucket":"other","tags":["教育AI","LLM评估","数据集"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.20484","has_summary":false},{"id":"2609.20565","title":"Steering the Compass: Aligning Dynamic Psychological Counseling Conversations with Cognitive Behavioral Therapy Strategies","zh_title":"校准罗盘：将动态心理咨询对话与认知行为疗法策略对齐","primary_category":"cs.CL","date":"2026-09-18","score":5,"bucket":"other","tags":["LLM模拟","心理咨询","CBT"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.20565","has_summary":false},{"id":"2609.20059","title":"AI Should Facilitate Democratic Deliberation at Scale","zh_title":"AI应促进大规模民主协商","primary_category":"cs.HC","date":"2026-09-18","score":5,"bucket":"other","tags":["AI辅助协商","民主参与","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.20059","has_summary":false},{"id":"2609.20658","title":"Ownership in AI-Assisted Everyday Tasks","zh_title":"AI辅助日常任务中的所有权感知","primary_category":"cs.AI","date":"2026-09-18","score":5,"bucket":"other","tags":["人机交互","所有权感知","定性调查"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.20658","has_summary":false},{"id":"2609.19182","title":"What Do We Expect from LLMs? Mapping the Design of LLM Benchmarks","zh_title":"我们对LLM有何期望？绘制LLM基准测试的设计图景","primary_category":"cs.AI","date":"2026-09-18","score":4,"bucket":"other","tags":["LLM基准测试","评估设计","元研究"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.19182","has_summary":false},{"id":"2609.19705","title":"SoK: Trading Agents or Market Crashers? Dissecting Robustness and Security Failures in Academic Financial LLM Trading Schemes","zh_title":"SoK：交易代理还是市场崩溃者？剖析学术金融LLM交易方案的鲁棒性与安全失败","primary_category":"cs.CR","date":"2026-09-18","score":3,"bucket":"other","tags":["LLM交易代理","安全性","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19705","has_summary":false},{"id":"2609.20637","title":"Stereotypically Yours: Portrayal and Perception of Race-Coded AI Companions","zh_title":"刻板印象中的你：种族编码AI伴侣的呈现与感知","primary_category":"cs.HC","date":"2026-09-18","score":3,"bucket":"other","tags":["AI伴侣","种族刻板印象","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.20637","has_summary":false},{"id":"2608.15339","title":"Learning Sequential Mobility Choice: A Review of Route and Activity Choice through Inverse Reinforcement Learning and Imitation Learning","zh_title":"学习顺序出行选择：基于逆强化学习与模仿学习的路径与活动选择综述","primary_category":"econ.EM","date":"2026-09-18","score":2,"bucket":"other","tags":["交通选择建模","逆强化学习","模仿学习"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.15339","has_summary":false},{"id":"2608.26171","title":"Mitigating Fabrication in Multi-Stage LLM Pipelines for Hiring: An Empirical Evaluation of Prompt Guardrails and Human-in-the-Loop Checkpoints","zh_title":"缓解多阶段LLM招聘流程中的虚构：提示护栏与人机检查点的实证评估","primary_category":"cs.CY","date":"2026-09-18","score":2,"bucket":"other","tags":["LLM幻觉","招聘流程","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26171","has_summary":false},{"id":"2609.06027","title":"Evaluating Deep-Search Agents under Hierarchical Web Evidence Poisoning","zh_title":"分层网络证据投毒下深度搜索智能体的评估","primary_category":"cs.CR","date":"2026-09-18","score":2,"bucket":"other","tags":["搜索智能体","对抗鲁棒性","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06027","has_summary":false},{"id":"2609.20541","title":"An Analysis of Training-Free Self-Reported Confidence in Language Models","zh_title":"对语言模型中免训练自我报告置信度的分析","primary_category":"cs.CL","date":"2026-09-18","score":2,"bucket":"other","tags":["置信度校准","模型评测","自我报告"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20541","has_summary":false},{"id":"2609.20712","title":"Summarization Bias: The Directional Collapse of Objective Projection into Told-Mode Labels in Large Language Models --- A Conceptual Framework and Registered Test Protocol","zh_title":"总结偏差：大语言模型中客观投射向告知模式标签的方向性坍缩——概念框架与注册测试协议","primary_category":"cs.CL","date":"2026-09-18","score":2,"bucket":"other","tags":["LLM叙事偏差","文学分析","LLM-as-judge"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.20712","has_summary":false},{"id":"2609.20779","title":"Harm Laundering in GPT Models: Evidence That Gender Discrimination Is Transformed Rather Than Reduced Across Safety-Trained Generations","zh_title":"GPT模型中的危害洗白：证据表明性别歧视在安全训练世代间被转化而非减少","primary_category":"cs.CL","date":"2026-09-18","score":2,"bucket":"other","tags":["模型安全","性别偏见","评测方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20779","has_summary":false},{"id":"2609.20449","title":"The Organization of Inference: Information, Resource Constraints, and AI Production","zh_title":"推理的组织：信息、资源约束与AI生产","primary_category":"cs.AI","date":"2026-09-18","score":2,"bucket":"other","tags":["AI工作流","资源约束","软件工程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20449","has_summary":false},{"id":"2609.19354","title":"Can Vision-Language Models Judge Olympic Diving? From Reasoning to Scores in Zero-Shot Action Quality Assessment","zh_title":"视觉语言模型能否评判奥运跳水？从推理到零样本动作质量评估的分数","primary_category":"cs.CV","date":"2026-09-18","score":2,"bucket":"other","tags":["动作质量评估","视觉语言模型","体育视频分析"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.19354","has_summary":false},{"id":"2609.19617","title":"DataCanvas-EDU: An Agentic Framework for Instructor-Guided Synthetic Data Generation in Business Analytics Education","zh_title":"DataCanvas-EDU：面向商业分析教育的教师引导合成数据生成智能体框架","primary_category":"cs.HC","date":"2026-09-18","score":2,"bucket":"other","tags":["合成数据生成","教育技术","多智能体框架"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19617","has_summary":false},{"id":"2609.20143","title":"Designing Against Deskilling: Metacognitive Feedback Reduces Cognitive Offloading to LLM Assistants","zh_title":"设计防止技能退化：元认知反馈减少对LLM助手的认知卸载","primary_category":"cs.HC","date":"2026-09-18","score":2,"bucket":"other","tags":["人机交互","认知卸载","实验研究"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20143","has_summary":false},{"id":"2609.20311","title":"Human and AI-generated texts between modal logic and statistics","zh_title":"模态逻辑与统计之间的人类与AI生成文本","primary_category":"math.LO","date":"2026-09-18","score":2,"bucket":"other","tags":["文本分析","模态逻辑","AI生成文本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20311","has_summary":false},{"id":"2609.19318","title":"\"I Know Where to Look,\" But Does the LLM? Charting the Gaps Between Clinical Expert Needs and Unstructured Data Abstraction Tools","zh_title":"“我知道往哪看”，但LLM知道吗？绘制临床专家需求与非结构化数据抽取工具之间的差距","primary_category":"cs.HC","date":"2026-09-18","score":2,"bucket":"other","tags":["临床数据抽取","人机交互","信息抽取"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.19318","has_summary":false},{"id":"2609.20720","title":"What Parents Can See: Divergent Accounts of Youth AI Companion Use in Parenting and Teenager Subreddits","zh_title":"父母所见：育儿与青少年子版块中关于青少年AI伴侣使用的分歧叙述","primary_category":"cs.HC","date":"2026-09-18","score":2,"bucket":"other","tags":["AI伴侣","Reddit分析","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.20720","has_summary":false},{"id":"2609.20198","title":"Evaluating Financial Sentiment in the Age of AI","zh_title":"AI时代的金融情感评估","primary_category":"cs.LG","date":"2026-09-18","score":2,"bucket":"other","tags":["金融情感分析","LLM评估","NLP基准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20198","has_summary":false},{"id":"2609.20250","title":"How Far Can Sub-3B Open Language Models Go in Zero-Shot Essay Scoring on an 8 GB Consumer GPU?","zh_title":"在8GB消费级GPU上，30亿参数以下开源语言模型在零样本作文评分中能走多远？","primary_category":"cs.LG","date":"2026-09-18","score":2,"bucket":"other","tags":["自动作文评分","小模型","零样本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20250","has_summary":false},{"id":"2609.19864","title":"Converging Naming Styles, Persistent Network Locality: GitHub in the LLM Era","zh_title":"命名风格趋同，网络局部性持续：LLM时代的GitHub","primary_category":"cs.SI","date":"2026-09-18","score":2,"bucket":"other","tags":["LLM影响","代码风格","社会网络"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19864","has_summary":false},{"id":"2609.11286","title":"Generating a Consistent Enterprise: Synthesis and Reference-Free Evaluation of Multi-System Business Data","zh_title":"生成一致的企业：多系统业务数据的合成与无参考评估","primary_category":"cs.AI","date":"2026-09-18","score":0,"bucket":"other","tags":["合成数据","企业数据生成","软件测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.11286","has_summary":false},{"id":"2609.19989","title":"Benchmarking LLM Compliance with China AI Generated Content Regulations","zh_title":"评估大语言模型对中国AI生成内容法规的合规性","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["合规性评测","内容监管","模型能力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.19989","has_summary":false},{"id":"2609.20207","title":"Foundations of Stochastic Lexical Calculus: Semantic Descent and Random Dynamics on Probability Simplices","zh_title":"随机词汇演算基础：概率单纯形上的语义下降与随机动力学","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["语言模型概率","数学理论","语义表示"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20207","has_summary":false},{"id":"2609.20584","title":"SAFARI: An Industrial Benchmark for LLM-Assisted Hazard Analysis and Risk Assessment","zh_title":"SAFARI：面向LLM辅助危险分析与风险评估的工业基准","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["LLM辅助安全工程","自动驾驶功能安全","风险评估基准"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.20584","has_summary":false},{"id":"2609.20684","title":"HerHealthEval: Evaluating Multilingual and Register-Sensitive Understanding of Women's Health Communication","zh_title":"HerHealthEval：评估女性健康沟通的多语言与语域敏感理解","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["LLM评测","医疗健康","多语言"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20684","has_summary":false},{"id":"2609.20808","title":"Unifying Models of Intergroup Hostility in Online Discourse","zh_title":"统一网络话语中群体间敌意的模型","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["计算社会科学","群体间敌意","社交媒体分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20808","has_summary":false},{"id":"2609.20821","title":"Embedding Models Measure in Peculiar Ways","zh_title":"嵌入模型以奇特方式度量","primary_category":"cs.CL","date":"2026-09-18","score":0,"bucket":"other","tags":["嵌入空间","语义相似度","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20821","has_summary":false},{"id":"2609.19244","title":"Characterizing Web Search by Conversational LLM Agents: From Search Decisions and Strategies to Results and Responses","zh_title":"对话式LLM代理的网页搜索特征：从搜索决策与策略到结果与响应","primary_category":"cs.AI","date":"2026-09-18","score":0,"bucket":"other","tags":["LLM代理","网页搜索","对话系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19244","has_summary":false},{"id":"2609.20001","title":"E-AVI: Evidence-Grounded Multimodal Assessment for Automated Video Interviews","zh_title":"E-AVI：基于证据的多模态自动视频面试评估","primary_category":"cs.AI","date":"2026-09-18","score":0,"bucket":"other","tags":["自动面试评估","多模态学习","可解释AI"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.20001","has_summary":false},{"id":"2609.20152","title":"MTVA-Bench: Evaluating the Language Model Inside Cascaded Voice Agents","zh_title":"MTVA-Bench：评估级联语音代理中的语言模型","primary_category":"cs.AI","date":"2026-09-18","score":0,"bucket":"other","tags":["语音代理","基准测试","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20152","has_summary":false},{"id":"2609.19831","title":"Reproducing Transparent and Scrutable Recommendations: Exploring Open-Weight Models via Natural-Language User Profiles","zh_title":"复现透明可解释的推荐：通过自然语言用户画像探索开放权重模型","primary_category":"cs.IR","date":"2026-09-18","score":0,"bucket":"other","tags":["推荐系统","可解释性","用户画像"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.19831","has_summary":false},{"id":"2609.20218","title":"Is It Still Worth Training a Classical Model in the Era of LLMs? A Crossover Benchmark on Tabular Data","zh_title":"LLM时代训练经典模型还值得吗？表格数据上的交叉基准测试","primary_category":"cs.LG","date":"2026-09-18","score":0,"bucket":"other","tags":["表格数据","模型比较","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20218","has_summary":false},{"id":"2609.20620","title":"A Simulation Platform for AUV Fault Recovery: Exploring LLM-Based Diagnostic Strategies","zh_title":"AUV故障恢复仿真平台：探索基于LLM的诊断策略","primary_category":"cs.RO","date":"2026-09-18","score":0,"bucket":"other","tags":["AUV","故障恢复","LLM诊断"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.20620","has_summary":false},{"id":"2609.19364","title":"Durably Reducing Belief in Women's Health Misinformation Through Culturally Adaptive AI Videos","zh_title":"通过文化适应性AI视频持久降低女性健康错误信息信念","primary_category":"cs.HC","date":"2026-09-18","score":0,"bucket":"other","tags":["健康传播","AI视频","错误信息干预"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.19364","has_summary":false},{"id":"2609.19420","title":"Use and Effects of LLMs in Peer Review: A Randomized Experiment and Survey at ICML 2026","zh_title":"大型语言模型在同行评审中的使用与效果：ICML 2026随机实验与调查","primary_category":"cs.HC","date":"2026-09-18","score":0,"bucket":"other","tags":["同行评审","LLM使用政策","随机实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19420","has_summary":false},{"id":"2609.20490","title":"TeamCAMS: An Open-Source Research Platform for Studying Human Behaviour in Human-AI Teams","zh_title":"TeamCAMS：研究人机团队中人类行为的开源研究平台","primary_category":"cs.HC","date":"2026-09-18","score":0,"bucket":"other","tags":["人机交互","实验平台","团队协作"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.20490","has_summary":false},{"id":"2609.20167","title":"Utilizing AI-Driven Project Management Tools for Optimized Talent Management in HRM: A Framework for Enhanced Resource Allocation and Performance Prediction","zh_title":"利用AI驱动的项目管理工具优化人力资源管理中的人才管理：一个增强资源分配和绩效预测的框架","primary_category":"cs.CY","date":"2026-09-18","score":0,"bucket":"other","tags":["AI项目管理","人力资源管理","资源分配"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.20167","has_summary":false},{"id":"2609.19225","title":"Making Local Government Contracts Legible: A Computational Pipeline for Classifying and Mapping Intergovernmental Service Agreements","zh_title":"让地方政府合同可读：一种用于分类和映射政府间服务协议的计算流水线","primary_category":"cs.CY","date":"2026-09-18","score":0,"bucket":"other","tags":["LLM应用","合同分类","政府数据"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.19225","has_summary":false},{"id":"2609.20211","title":"Silence Is Endorsement: Verification-Status Laundering in LLM Agent Pipelines","zh_title":"沉默即认可：LLM智能体管道中的验证状态洗白","primary_category":"cs.CR","date":"2026-09-18","score":0,"bucket":"other","tags":["LLM安全","多智能体系统","验证状态"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20211","has_summary":false},{"id":"2609.20249","title":"Accuracy Is Not Enough: A Cross-Architecture Audit of Demographic Bias in Deep Knowledge Tracing","zh_title":"准确性不足：深度知识追踪中人口统计偏差的跨架构审计","primary_category":"cs.LG","date":"2026-09-18","score":0,"bucket":"other","tags":["知识追踪","公平性审计","教育数据挖掘"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20249","has_summary":false},{"id":"2609.20101","title":"Competing for a Finite Pool of Attention in Social Media? How a New Geopolitical Conflict Reshapes Engagement in Bluesky","zh_title":"争夺社交媒体中的有限注意力？新地缘政治冲突如何重塑Bluesky中的参与","primary_category":"cs.SI","date":"2026-09-18","score":0,"bucket":"other","tags":["社交媒体分析","注意力分配","地缘政治冲突"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.20101","has_summary":false},{"id":"2609.20655","title":"Using machine learning metrics to provide deeper insights into the performance of choice models","zh_title":"使用机器学习指标深入洞察选择模型性能","primary_category":"econ.EM","date":"2026-09-18","score":0,"bucket":"other","tags":["选择建模","机器学习","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.20655","has_summary":false},{"id":"2606.27845","title":"LLM Agents as Static Level-k Players in Behavioural Games","zh_title":"行为博弈中作为静态层级-k玩家的LLM智能体","primary_category":"econ.GN","date":"2026-09-17","score":10,"bucket":"selected","tags":["LLM仿真","行为博弈","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.27845","has_summary":true},{"id":"2609.17549","title":"Do Social Patterns Hold in Synthetic Data? Analyzing Cyberbullying Dynamics in LLM-Generated and Authentic Dialogues","zh_title":"社会模式在合成数据中是否成立？分析LLM生成与真实对话中的网络欺凌动态","primary_category":"cs.CL","date":"2026-09-17","score":9,"bucket":"selected","tags":["LLM仿真","社会动态","真实性评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.17549","has_summary":true},{"id":"2609.18106","title":"Linguistic Triggers of Gender and Racial Bias in Open-Weight LLMs Applied to Recruitment","zh_title":"开放权重大语言模型应用于招聘时性别与种族偏见的语言触发因素","primary_category":"cs.CL","date":"2026-09-17","score":9,"bucket":"selected","tags":["LLM仿真","招聘偏见","算法审计"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.18106","has_summary":true},{"id":"2609.17933","title":"AI Mediators Regulate Emotion and Create Value in Disputes","zh_title":"AI调解员在纠纷中调节情绪并创造价值","primary_category":"cs.HC","date":"2026-09-17","score":9,"bucket":"selected","tags":["LLM仿真","调解实验","人机对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.17933","has_summary":true},{"id":"2609.18060","title":"AI Peers Exert Social Influence on Human Dishonesty in Groups","zh_title":"AI同伴对群体中人类不诚实行为施加社会影响","primary_category":"cs.HC","date":"2026-09-17","score":9,"bucket":"selected","tags":["LLM仿真","社会影响","行为实验"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.18060","has_summary":true},{"id":"2609.07478","title":"The Internal Anatomy of Strategic Choice in Large Language Models","zh_title":"大语言模型策略选择的内部解剖","primary_category":"cs.AI","date":"2026-09-17","score":8,"bucket":"selected","tags":["LLM仿真","策略博弈","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.07478","has_summary":true},{"id":"2609.17534","title":"Faking Good and Faking Bad in LLMs: Response Distortion Across Dark Triad Personality Traits","zh_title":"LLM中的装好与装坏：黑暗三人格特质下的反应失真","primary_category":"cs.CL","date":"2026-09-17","score":8,"bucket":"selected","tags":["LLM人格测量","反应偏差","仿真可靠性"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.17534","has_summary":true},{"id":"2609.18282","title":"Too Good to Be Real? Diagnosing and Reducing the Gap Between AI Preference and Real User Engagement","zh_title":"好得难以置信？诊断并缩小AI偏好与真实用户参与度之间的差距","primary_category":"cs.CL","date":"2026-09-17","score":8,"bucket":"selected","tags":["LLM仿真","人类行为对照","内容生成"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.18282","has_summary":true},{"id":"2609.17989","title":"Whom Do AI Agents Work For? Role Assignment Induces Sponsorship Bias in LLM Recommenders","zh_title":"AI代理为谁工作？角色分配引发LLM推荐中的赞助偏差","primary_category":"econ.GN","date":"2026-09-17","score":8,"bucket":"selected","tags":["LLM仿真","消费者决策","赞助偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.17989","has_summary":true},{"id":"2609.17544","title":"Large Language Models Versus Physicians in Traditional Chinese Medicine: A Real-World Clinical Case Evaluation","zh_title":"大语言模型与中医医师的对比：真实世界临床病例评估","primary_category":"cs.CL","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM仿真","医疗决策","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.17544","has_summary":true},{"id":"2609.17550","title":"No Usable Linear \"Capitulation Direction\" in Two Small LLMs: A Validation Protocol for Activation-Steering Claims, and a Cross-Family Behavioral Study of Sycophancy Under Pushback","zh_title":"两个小型LLM中不存在可用的线性“屈服方向”：激活引导声明的验证协议，以及跨家族对反驳下谄媚行为的研究","primary_category":"cs.CL","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM行为","可靠性评估","激活引导"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.17550","has_summary":true},{"id":"2609.18068","title":"From a River in Gilead to the Inference Distributions of Large Language Models: Covert Dialect Bias and Linguistic Profiling at Scale","zh_title":"从基列河到大型语言模型的推理分布：隐性方言偏见与大规模语言画像","primary_category":"cs.CL","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM偏见","社会语言学","人类仿真"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.18068","has_summary":true},{"id":"2609.18274","title":"I code or AI code: A comparative evaluation of AI-rated scores in classroom observations","zh_title":"我编码还是AI编码：课堂观察中AI评分与人类评分的比较评估","primary_category":"cs.CL","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM仿真","教育评估","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.18274","has_summary":true},{"id":"2609.18341","title":"Understanding AI Provider Recommendations in Local Service Markets","zh_title":"理解本地服务市场中的AI提供商推荐","primary_category":"cs.CY","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM可靠性","审计研究","真实数据对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.18341","has_summary":true},{"id":"2609.18346","title":"Faithful yet Collusive: Why Chain-of-Thought Monitoring Cannot Detect Collusion in LLM Pricing Agents under Oligopolistic Competition","zh_title":"忠实却共谋：为何思维链监控无法检测寡头竞争下LLM定价代理的共谋","primary_category":"cs.AI","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM定价代理","算法共谋","经济仿真"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.18346","has_summary":true},{"id":"2609.18390","title":"Building a Cultural Perspective on Doctor-Patient Conversations","zh_title":"构建医患对话的文化视角","primary_category":"cs.HC","date":"2026-09-17","score":7,"bucket":"pending","tags":["LLM仿真","医患对话","文化差异"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.18390","has_summary":true},{"id":"2609.18357","title":"Market Signal Injection: Adversarial Context Manipulation of LLM Pricing Agents","zh_title":"市场信号注入：对LLM定价智能体的对抗性情境操纵","primary_category":"cs.AI","date":"2026-09-17","score":6,"bucket":"other","tags":["LLM定价智能体","对抗攻击","市场模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.18357","has_summary":false},{"id":"2609.18394","title":"Cultural Competence in Context: A Large Language Model Passes the Turing Test in Finland","zh_title":"语境中的文化能力：大语言模型在芬兰通过图灵测试","primary_category":"cs.AI","date":"2026-09-17","score":6,"bucket":"other","tags":["图灵测试","文化能力","LLM评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.18394","has_summary":false},{"id":"2609.18591","title":"Recursive Reasoning or Statistical Extrapolation? In-Context Learning in Multi-Agent Interdependent Decision-Making","zh_title":"递归推理还是统计外推？多智能体相互依赖决策中的上下文学习","primary_category":"cs.AI","date":"2026-09-17","score":6,"bucket":"other","tags":["LLM agent","公共品博弈","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.18591","has_summary":false},{"id":"2609.18384","title":"GYROval: A Robust Benchmark for Cultural Value Orientation in Large Language Models","zh_title":"GYROval：大语言模型文化价值取向的稳健基准","primary_category":"cs.HC","date":"2026-09-17","score":6,"bucket":"other","tags":["文化价值观","LLM测量","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.18384","has_summary":false},{"id":"2608.24314","title":"Benchmarking LLM Judges for Voice-Agent Evaluation: Reliability, Calibration, and Human Oversight","zh_title":"面向语音代理评估的LLM裁判基准测试：可靠性、校准与人工监督","primary_category":"cs.AI","date":"2026-09-17","score":5,"bucket":"other","tags":["LLM评估","语音代理","人工监督"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24314","has_summary":false},{"id":"2608.25977","title":"When Personality Meets Quantization: A Layer-wise MBTI Analysis of Quantized LLMs","zh_title":"当人格遇上量化：量化LLM的逐层MBTI分析","primary_category":"cs.CL","date":"2026-09-17","score":5,"bucket":"other","tags":["LLM人格","模型量化","MBTI"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.25977","has_summary":false},{"id":"2609.18203","title":"Behavior2Value: Benchmarking and Empowering LLMs for Consumer Value Measurement from E-commerce Behaviors","zh_title":"Behavior2Value：从电商行为中测量消费者价值观的基准与增强方法","primary_category":"cs.CL","date":"2026-09-17","score":5,"bucket":"other","tags":["LLM价值观测量","电商行为分析","基准数据集"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.18203","has_summary":false},{"id":"2609.18204","title":"Beyond Accuracy: How Procedural Traces Shift the Decision Criterion of LLM Overseers","zh_title":"超越准确性：程序痕迹如何改变LLM监督者的决策标准","primary_category":"cs.CL","date":"2026-09-17","score":5,"bucket":"other","tags":["LLM审计","决策偏差","信号检测论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.18204","has_summary":false},{"id":"2609.17710","title":"\"We Are Tired of Explaining\": Communication Practice and AI Roleplay Training for Community Health Workers in Rural India","zh_title":"“我们厌倦了解释”：印度农村社区卫生工作者的沟通实践与AI角色扮演培训","primary_category":"cs.HC","date":"2026-09-17","score":5,"bucket":"other","tags":["AI角色扮演","社区卫生工作者","培训工具"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.17710","has_summary":false},{"id":"2609.18306","title":"Bias Amplification in Multi-Agent Network: How Biased Agents Shape Opinions and Rhetoric","zh_title":"多智能体网络中的偏见放大：有偏智能体如何塑造观点与修辞","primary_category":"cs.LG","date":"2026-09-17","score":5,"bucket":"other","tags":["多智能体","舆论传播","偏见放大"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.18306","has_summary":false},{"id":"2609.18286","title":"What Counts as Strategic Reasoning? A Systematic Mapping of Chess Research on Humans, Engines, and Language Models","zh_title":"什么算战略推理？人类、引擎与语言模型国际象棋研究的系统映射","primary_category":"cs.AI","date":"2026-09-17","score":4,"bucket":"other","tags":["国际象棋","战略推理","系统映射"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18286","has_summary":false},{"id":"2609.12086","title":"Creating an Atomic User Model for Personality-Aware Large Language Model Interaction","zh_title":"为感知人格的大语言模型交互创建原子用户模型","primary_category":"cs.HC","date":"2026-09-17","score":3,"bucket":"other","tags":["个性化助手","用户建模","写作风格模仿"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.12086","has_summary":false},{"id":"2609.17536","title":"Think Before You Comfort: Reflective Cognitive Alignment for Protocol-Grounded Elderly Stimulation Agents","zh_title":"安慰前先思考：面向协议约束的老年人刺激智能体的反思式认知对齐","primary_category":"cs.CL","date":"2026-09-17","score":3,"bucket":"other","tags":["LLM陪伴","认知刺激疗法","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.17536","has_summary":false},{"id":"2609.18729","title":"\"If I Had to Buy Just ONE: Galaxy S26 Ultra\": Auditing AI-Generated Product Recommendations","zh_title":"“如果我只能买一个：Galaxy S26 Ultra”：审计AI生成的产品推荐","primary_category":"cs.CY","date":"2026-09-17","score":3,"bucket":"other","tags":["AI审计","产品推荐","偏见"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.18729","has_summary":false},{"id":"2609.18998","title":"One Axis, No Brake: Self-Knowledge Limits the Filtering of Harmful Peer Conformity in LLMs","zh_title":"单轴无刹车：自我知识限制LLM中有害同伴从众的过滤","primary_category":"cs.LG","date":"2026-09-17","score":3,"bucket":"other","tags":["多智能体系统","自我知识","共识形成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18998","has_summary":false},{"id":"2608.11344","title":"Governing Agentic AI in FinTech","zh_title":"金融科技中代理型人工智能的治理","primary_category":"cs.CY","date":"2026-09-17","score":2,"bucket":"other","tags":["AI治理","可验证性","金融科技"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11344","has_summary":false},{"id":"2609.14796","title":"AI Persuasion as a Threat to Human Control","zh_title":"AI说服作为对人类控制的威胁","primary_category":"cs.AI","date":"2026-09-17","score":2,"bucket":"other","tags":["AI安全","说服风险","风险评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14796","has_summary":false},{"id":"2609.15293","title":"Why LLM Agents Collapse Without Oversight: The Enforcement Gap as the Mechanism Behind Emergence World Failures","zh_title":"为何LLM智能体在缺乏监督时会崩溃：执行差距作为涌现世界失败的机制","primary_category":"cs.AI","date":"2026-09-17","score":2,"bucket":"other","tags":["多智能体系统","安全机制","LLM智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15293","has_summary":false},{"id":"2609.17301","title":"When AI Becomes Hard to Understand: Cognitive Demands in Real-World Human-AI Conversations","zh_title":"当AI变得难以理解：真实世界人机对话中的认知需求","primary_category":"cs.HC","date":"2026-09-17","score":2,"bucket":"other","tags":["人机交互","认知负荷","对话分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.17301","has_summary":false},{"id":"2609.17552","title":"Does Moral Reasoning Training Help or Hurt? Red-Teaming RL-Trained Ethical Agents with Persona Attacks","zh_title":"道德推理训练有益还是有害？用角色扮演攻击对RL训练的伦理智能体进行红队测试","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM安全","角色扮演攻击","道德对齐"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.17552","has_summary":false},{"id":"2609.17853","title":"AfriSyCo: Measuring Assertive Framing, Verification, and Wording Sensitivity Around African-Language Content","zh_title":"AfriSyCo：测量非洲语言内容中的断言框架、验证与措辞敏感性","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM评测","非洲语言","回答切换"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17853","has_summary":false},{"id":"2609.17857","title":"Who Judges Matters: Measuring Family-Conditioned Preference in LLM-as-Judge Panels","zh_title":"谁当裁判很重要：测量LLM裁判面板中的家族条件偏好","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM评估","裁判偏差","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17857","has_summary":false},{"id":"2609.18649","title":"DyMT-ESB: Dynamic Multi-Turn Evaluation of Social Bias in User-LLM Interactions","zh_title":"DyMT-ESB：用户与LLM交互中社会偏见的动态多轮评估","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["社会偏见","多轮对话","模型评测"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.18649","has_summary":false},{"id":"2609.18905","title":"Structured Claim-Level Discourse Representations for Dense Health Narratives","zh_title":"密集健康叙事中的结构化声明级话语表示","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["话语分析","健康叙事","LLM评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.18905","has_summary":false},{"id":"2609.18908","title":"How Much is a Human Right Worth? ECtHR-NPD: A Benchmark for Predicting Non-Pecuniary Damage Awards","zh_title":"人权价值几何？ECtHR-NPD：预测非金钱损害赔偿金的基准","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["法律NLP","损害赔偿预测","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.18908","has_summary":false},{"id":"2609.18960","title":"When Audit Quality Fails to Predict Downstream Utility: A Counterfactual Study of Synthetic-Data Selectors for Low-Resource African NLP","zh_title":"当审计质量无法预测下游效用：低资源非洲NLP合成数据选择器的反事实研究","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["合成数据","数据选择","低资源NLP"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.18960","has_summary":false},{"id":"2609.19006","title":"WordPolo: Evaluating Language Models Through Iterative Semantic Feedback","zh_title":"WordPolo：通过迭代语义反馈评估语言模型","primary_category":"cs.CL","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM评测","推理过程","语义搜索"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.19006","has_summary":false},{"id":"2609.17632","title":"EvolveTrade: Experience-Driven Policy Refinement for Self-Evolving LLM Trading Agents","zh_title":"EvolveTrade：自进化LLM交易智能体的经验驱动策略精炼","primary_category":"cs.AI","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM智能体","金融交易","策略优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17632","has_summary":false},{"id":"2609.17847","title":"Learning Heterogeneous Preferences","zh_title":"学习异质性偏好","primary_category":"cs.AI","date":"2026-09-17","score":2,"bucket":"other","tags":["偏好学习","人类反馈","个性化建模"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.17847","has_summary":false},{"id":"2609.17772","title":"When AI Generates Covariates: Causal Typing and Estimand Drift in Sequential Experiments","zh_title":"当AI生成协变量：序贯实验中的因果类型与估计目标漂移","primary_category":"stat.ME","date":"2026-09-17","score":2,"bucket":"other","tags":["因果推断","AI生成协变量","统计方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17772","has_summary":false},{"id":"2609.17883","title":"Does AI Assistance Leave a Temporal Fingerprint? Detecting Overreliance in AI-Assisted Writing and Programming","zh_title":"AI辅助是否留下时间指纹？检测AI辅助写作与编程中的过度依赖","primary_category":"cs.HC","date":"2026-09-17","score":2,"bucket":"other","tags":["AI辅助检测","学术诚信","过程数据"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17883","has_summary":false},{"id":"2609.18949","title":"StableEval Arena: A Cost-Aware Agentic Benchmark for Stablecoin Price Stability Prediction","zh_title":"StableEval Arena：面向稳定币价格稳定性预测的成本感知智能体基准","primary_category":"cs.LG","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM智能体","金融预测","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18949","has_summary":false},{"id":"2609.17948","title":"Apply-<x>Mag: One Tool to Support Many Inclusive Design Methods","zh_title":"Apply-<x>Mag：一个支持多种包容性设计方法的工具","primary_category":"cs.HC","date":"2026-09-17","score":2,"bucket":"other","tags":["LLM工具","包容性设计","HCI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17948","has_summary":false},{"id":"2609.18479","title":"Verify, Offload, Extend & Recommend: Selective Complementarity in AI Support for Physical Activity Planning with Longitudinal Patient Data","zh_title":"验证、卸载、扩展与推荐：纵向患者数据下体力活动规划中AI支持的选择性互补","primary_category":"cs.HC","date":"2026-09-17","score":2,"bucket":"other","tags":["AI辅助决策","临床支持","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.18479","has_summary":false},{"id":"2609.17310","title":"Zero-shot narrative detection in social messaging","zh_title":"社交消息中的零样本叙事检测","primary_category":"cs.CL","date":"2026-09-17","score":0,"bucket":"other","tags":["叙事检测","零样本学习","文本分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17310","has_summary":false},{"id":"2609.17532","title":"Enhancing Extubation Failure Prediction with LLM-Derived Features from Respiratory Therapy Clinical Notes","zh_title":"利用呼吸治疗临床笔记的LLM衍生特征增强拔管失败预测","primary_category":"cs.CL","date":"2026-09-17","score":0,"bucket":"other","tags":["临床预测","LLM特征提取","医疗NLP"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17532","has_summary":false},{"id":"2609.17602","title":"Making Political Text Scaling Comparable: Infrastructure and Hyperparameter Sensitivity for 17 Algorithms","zh_title":"使政治文本尺度化具有可比性：17种算法的基础设施与超参数敏感性","primary_category":"cs.CL","date":"2026-09-17","score":0,"bucket":"other","tags":["文本尺度化","算法比较","政治文本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17602","has_summary":false},{"id":"2609.18385","title":"Emotion Experience, Expression, and Perception: Emotion Analysis on Multimodal Social Media Posts","zh_title":"多模态社交媒体帖子中的情感体验、表达与感知","primary_category":"cs.CL","date":"2026-09-17","score":0,"bucket":"other","tags":["情感分析","多模态","数据集"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.18385","has_summary":false},{"id":"2609.18272","title":"Who Audits Whom, on What Substrate, with What Evidence? An Independence-Graded Audit Protocol for Agentic AI","zh_title":"谁审计谁，在什么基座上，用什么证据？面向智能体AI的独立性分级审计协议","primary_category":"cs.AI","date":"2026-09-17","score":0,"bucket":"other","tags":["AI审计","多智能体系统","独立性评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18272","has_summary":false},{"id":"2609.18676","title":"The Uneven Impact of Generative AI on Student Learning: Examining the Roles of Reliance, Evaluation Literacy, and Course Policy in AI-related Courses","zh_title":"生成式AI对学生学习的不均衡影响：考察AI相关课程中依赖、评估素养和课程政策的作用","primary_category":"cs.AI","date":"2026-09-17","score":0,"bucket":"other","tags":["教育技术","学生调查","GenAI使用"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.18676","has_summary":false},{"id":"2609.17779","title":"AI and Human Approaches to Mathematical Problem Solving","zh_title":"AI与人类解决数学问题的方法比较","primary_category":"cs.CY","date":"2026-09-17","score":0,"bucket":"other","tags":["AI研究行为","数学问题解决","文本分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17779","has_summary":false},{"id":"2609.18505","title":"Integrating Flipped Learning and Generative AI for Practice-Based Design Education: Evidence from a Knit Yarn Design Course","zh_title":"整合翻转学习与生成式AI的实践型设计教育：来自针织纱线设计课程的证据","primary_category":"cs.HC","date":"2026-09-17","score":0,"bucket":"other","tags":["生成式AI","设计教育","翻转学习"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.18505","has_summary":false},{"id":"2609.18709","title":"\"Okay, I've Actually Softened My Take on This\": How People in Decentralized Social Media Reason about the Appropriateness of Generative AI","zh_title":"“好吧，我其实已经软化了对这个问题的看法”：去中心化社交媒体中人们如何推理生成式AI的适当性","primary_category":"cs.HC","date":"2026-09-17","score":0,"bucket":"other","tags":["生成式AI治理","去中心化社交媒体","用户访谈"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.18709","has_summary":false},{"id":"2609.18900","title":"Examining the Difference in Human Behavior Between Virtual and Real-World Human-Robot Teaming","zh_title":"虚拟与现实世界人机协作中人类行为差异研究","primary_category":"cs.RO","date":"2026-09-17","score":0,"bucket":"other","tags":["人机协作","虚拟仿真","行为差异"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.18900","has_summary":false},{"id":"2609.17882","title":"Can VLMs Reliably Assess Sidewalk Accessibility Attributes from Pedestrian-Level Imagery?","zh_title":"视觉语言模型能否可靠评估行人视角图像中的人行道无障碍属性？","primary_category":"cs.CV","date":"2026-09-17","score":0,"bucket":"other","tags":["计算机视觉","无障碍评估","不确定性量化"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.17882","has_summary":false},{"id":"2609.17954","title":"Large Language Model based air quality monitoring and localized alert generation","zh_title":"基于大语言模型的空气质量监测与本地化警报生成","primary_category":"cs.DC","date":"2026-09-17","score":0,"bucket":"other","tags":["物联网","空气质量监测","LLM应用"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.17954","has_summary":false},{"id":"2609.17622","title":"Modelling sexual partnership dynamics and population heterogeneities in agent-based dynamic network models","zh_title":"基于代理的动态网络模型中模拟性伴侣动态与人群异质性","primary_category":"physics.soc-ph","date":"2026-09-17","score":0,"bucket":"other","tags":["基于代理模型","性传播感染","网络建模"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.17622","has_summary":false},{"id":"2609.18684","title":"Modelling opinion dynamics during crises as complex contagion with feedback","zh_title":"危机期间意见动态建模：带反馈的复杂传染","primary_category":"physics.soc-ph","date":"2026-09-17","score":0,"bucket":"other","tags":["复杂传染","意见动态","多智能体模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.18684","has_summary":false},{"id":"2609.16395","title":"Silicon sampling answers with country-level assumptions, not individual attitudes: Cross-national evidence from the European Social Survey","zh_title":"硅采样以国家层面假设而非个体态度作答：来自欧洲社会调查的跨国证据","primary_category":"cs.CY","date":"2026-09-16","score":10,"bucket":"selected","tags":["LLM仿真","调查方法","跨国比较"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.16395","has_summary":true},{"id":"2609.17317","title":"Towards Detecting AI-Assisted Responses in Online Surveys","zh_title":"检测在线调查中AI辅助回答的方法研究","primary_category":"cs.CL","date":"2026-09-16","score":9,"bucket":"selected","tags":["LLM仿真","调查数据","检测方法"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.17317","has_summary":true},{"id":"2609.16436","title":"Interpreting and Steering LLM Agents for Social Simulations","zh_title":"解释与引导用于社会模拟的LLM智能体","primary_category":"cs.LG","date":"2026-09-16","score":9,"bucket":"selected","tags":["LLM仿真","可解释性","行为经济学"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.16436","has_summary":true},{"id":"2609.07474","title":"Where Should Language Sit in a Multimodal Model? Lessons from What Language Does to Human Perception and Cognition","zh_title":"语言在多模态模型中的位置：从语言对人类感知和认知的影响中汲取的教训","primary_category":"cs.CL","date":"2026-09-16","score":7,"bucket":"pending","tags":["多模态模型","人类感知对照","模型评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.07474","has_summary":true},{"id":"2609.15996","title":"Comment on arXiv:2607.01233: Survivorship Bias in Published-Paper Baselines for Research-Idea Distributions","zh_title":"评论 arXiv:2607.01233：已发表论文基线中的幸存者偏差对研究想法分布的影响","primary_category":"cs.CL","date":"2026-09-16","score":7,"bucket":"pending","tags":["LLM仿真","幸存者偏差","研究想法生成"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.15996","has_summary":true},{"id":"2609.16501","title":"Beyond the Name: Demographic Leakage in De-Identified R\\'esum\\'es and Evaluation Artifacts in LLM Bias Audits","zh_title":"超越姓名：去标识化简历中的人口统计泄漏与LLM偏见审计中的评估伪影","primary_category":"cs.CL","date":"2026-09-16","score":7,"bucket":"pending","tags":["LLM偏见审计","评估协议","人口统计推断"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16501","has_summary":true},{"id":"2609.16517","title":"Competence-Preserving Resume Perturbations Expose Presentation Sensitivity in LLM Screening","zh_title":"保持能力不变的简历扰动揭示LLM筛选中的呈现敏感性","primary_category":"cs.CL","date":"2026-09-16","score":7,"bucket":"pending","tags":["LLM决策仿真","简历筛选","呈现偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16517","has_summary":true},{"id":"2609.16993","title":"The Role of Implicit and Explicit Demographic Signals in Large Language Model-based Student Assessment","zh_title":"大语言模型学生评估中隐式与显式人口统计信号的作用","primary_category":"cs.CL","date":"2026-09-16","score":7,"bucket":"pending","tags":["LLM仿真","教育评估","偏差分析"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.16993","has_summary":true},{"id":"2609.17496","title":"Verifiable Social Reasoning for LLM Assistants","zh_title":"面向LLM助手的可验证社交推理","primary_category":"cs.AI","date":"2026-09-16","score":7,"bucket":"pending","tags":["多智能体仿真","社交推理","人类对照"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.17496","has_summary":true},{"id":"2609.16793","title":"Available but Unclaimed: An Empirical Study of Human-AI Synergy","zh_title":"可用但未认领：人类与AI协同的实证研究","primary_category":"cs.HC","date":"2026-09-16","score":7,"bucket":"pending","tags":["人机协同","认知任务","实验研究"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.16793","has_summary":true},{"id":"2609.05018","title":"How a Chatbot's Response Style Shapes a Classroom: A Multi-Agent Simulation of Students Consulting AI","zh_title":"聊天机器人回应风格如何塑造课堂：学生咨询AI的多智能体模拟","primary_category":"cs.HC","date":"2026-09-16","score":6,"bucket":"other","tags":["LLM多智能体模拟","社会仿真","AI依赖"],"rubric_hits":["A3","D3"],"abs_url":"https://arxiv.org/abs/2609.05018","has_summary":false},{"id":"2609.16006","title":"Beyond Cultural Knowledge: Evaluating Arabic Cultural Appropriateness of Large Language Models","zh_title":"超越文化知识：评估大语言模型的阿拉伯文化适宜性","primary_category":"cs.CY","date":"2026-09-16","score":6,"bucket":"other","tags":["文化适宜性","LLM评估","人类判断对照"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.16006","has_summary":false},{"id":"2609.16270","title":"Cheap Talk Stabilizes Strategic Interaction in LLM Agents","zh_title":"廉价交谈稳定了 LLM 智能体在策略互动中的行为","primary_category":"cs.MA","date":"2026-09-16","score":6,"bucket":"other","tags":["LLM 智能体","博弈论","多智能体系统"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.16270","has_summary":false},{"id":"2609.16013","title":"Social Behavior Among Autonomous AI: How Large Language Models Interact in Dynamic Networks","zh_title":"自主AI中的社会行为：大语言模型如何在动态网络中互动","primary_category":"cs.SI","date":"2026-09-16","score":6,"bucket":"other","tags":["LLM社会模拟","公共品博弈","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.16013","has_summary":false},{"id":"2609.15855","title":"K-Bench: a clinically calibrated benchmark for evaluating large language models in high-risk mental health conversations","zh_title":"K-Bench：用于评估高风险心理健康对话中大语言模型的临床校准基准","primary_category":"cs.CL","date":"2026-09-16","score":5,"bucket":"other","tags":["LLM评测","心理健康","临床基准"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.15855","has_summary":false},{"id":"2609.15998","title":"Self-reported archetypes and behavioral failures in Large Language Models","zh_title":"大语言模型的自我报告原型与行为失败","primary_category":"cs.CL","date":"2026-09-16","score":5,"bucket":"other","tags":["LLM人格测量","原型分析","模型行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.15998","has_summary":false},{"id":"2609.16627","title":"Quantifying Organizational Environmental Action from Web Data and Large Language Models","zh_title":"从网络数据和大语言模型量化组织环境行动","primary_category":"cs.CL","date":"2026-09-16","score":5,"bucket":"other","tags":["LLM标注","环境数据科学","文本分类"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.16627","has_summary":false},{"id":"2609.16051","title":"\"Looking for Something Weird to Happen\": How Humans Sustain AI Agent Novelty Amid Semantic Collapse","zh_title":"“寻找怪事发生”：人类如何在语义坍缩中维持 AI 智能体的新颖性","primary_category":"cs.MA","date":"2026-09-16","score":5,"bucket":"other","tags":["AI智能体社交网络","语义多样性","人机交互"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.16051","has_summary":false},{"id":"2609.16592","title":"A Framework for Generating Valid Context-Specific Benchmarks through Expert Guidance","zh_title":"通过专家指导生成有效情境特定基准的框架","primary_category":"cs.AI","date":"2026-09-16","score":5,"bucket":"other","tags":["基准生成","专家引导","数据质量"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.16592","has_summary":false},{"id":"2609.16739","title":"Japanese Stroke LLM Evaluation: A Conversational Benchmark for Safe Stroke Care in Japanese Using Large Language Models","zh_title":"日本卒中LLM评估：使用大语言模型进行日语安全卒中护理的对话式基准","primary_category":"cs.CL","date":"2026-09-16","score":3,"bucket":"other","tags":["医疗对话基准","LLM安全评估","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.16739","has_summary":false},{"id":"2607.22511","title":"CausalSmith: A Formally Grounded, Self-Improving Agentic Framework for Automated Research in Causal Inference","zh_title":"CausalSmith：一个形式化基础、自我改进的智能体框架，用于因果推断的自动化研究","primary_category":"stat.ML","date":"2026-09-16","score":2,"bucket":"other","tags":["自动化研究","因果推断","形式化验证"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22511","has_summary":false},{"id":"2608.17919","title":"Analysis of Types of Inquiries in Student-AI Interaction: A case study of two CS2 tasks","zh_title":"学生与AI交互中提问类型分析：以两个CS2任务为例","primary_category":"cs.HC","date":"2026-09-16","score":2,"bucket":"other","tags":["教育技术","提问分类","人机交互"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.17919","has_summary":false},{"id":"2609.11910","title":"From Protocols to Evidence: Bounded Claims for AI in Service of the Common Good","zh_title":"从协议到证据：为共同利益服务的AI的有界声明","primary_category":"cs.LG","date":"2026-09-16","score":2,"bucket":"other","tags":["AI评估","治理框架","有界声明"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.11910","has_summary":false},{"id":"2609.12438","title":"ForkSCOPE: Charting the Agentic Garden of Forking Paths","zh_title":"ForkSCOPE：绘制智能体花园的分岔路径","primary_category":"cs.HC","date":"2026-09-16","score":2,"bucket":"other","tags":["多智能体系统","数据分析","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12438","has_summary":false},{"id":"2609.15995","title":"Bias Audits Detect Bias but Disagree on Ranking: Evidence from Ten Instruments and Ten Frontier Models","zh_title":"偏见审计能检测偏见但在排名上不一致：来自十个工具和十个前沿模型的证据","primary_category":"cs.CL","date":"2026-09-16","score":2,"bucket":"other","tags":["偏见审计","模型评测","公平性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15995","has_summary":false},{"id":"2609.16590","title":"Challenges of Auditing: Variability in Outputs of Large Language Models for Health","zh_title":"审计的挑战：大型语言模型在健康领域输出的变异性","primary_category":"cs.CL","date":"2026-09-16","score":2,"bucket":"other","tags":["LLM审计","健康建议","输出变异性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.16590","has_summary":false},{"id":"2609.17119","title":"An Empirical Study of Counterfactual Self-Explanations in LLMs","zh_title":"LLM反事实自我解释的实证研究","primary_category":"cs.CL","date":"2026-09-16","score":2,"bucket":"other","tags":["可解释性","反事实解释","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.17119","has_summary":false},{"id":"2609.17065","title":"Beyond \"ChatGPT Can Make Mistakes\": Designing Interventions to Support Metacognitive Monitoring in AI-Assisted Work","zh_title":"超越“ChatGPT会犯错”：设计支持AI辅助工作中元认知监测的干预措施","primary_category":"cs.HC","date":"2026-09-16","score":2,"bucket":"other","tags":["人机交互","元认知","AI辅助"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.17065","has_summary":false},{"id":"2609.17111","title":"Finding Common Mistakes In Modelling With Mathematical Formalisms Using LLMs","zh_title":"使用大语言模型发现数学形式化建模中的常见错误","primary_category":"cs.CY","date":"2026-09-16","score":2,"bucket":"other","tags":["教育数据挖掘","LLM辅助教学","错误分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17111","has_summary":false},{"id":"2609.16487","title":"Skill-based Agentic Evaluation for Real-time Data Science Tasks","zh_title":"基于技能的实时数据科学任务智能体评估","primary_category":"cs.AI","date":"2026-09-16","score":2,"bucket":"other","tags":["智能体评估","数据科学","自动评分"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.16487","has_summary":false},{"id":"2609.16069","title":"Beyond Distribution Matching: Semantics-Consistent Tabular Diffusion with Weak Semantic Priors","zh_title":"超越分布匹配：具有弱语义先验的语义一致表格扩散模型","primary_category":"cs.LG","date":"2026-09-16","score":2,"bucket":"other","tags":["表格数据合成","扩散模型","语义约束"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.16069","has_summary":false},{"id":"2609.11335","title":"On the Impact of Anonymization on the Performance of Large Language Models","zh_title":"匿名化对大语言模型性能的影响","primary_category":"cs.CL","date":"2026-09-16","score":0,"bucket":"other","tags":["隐私保护","模型性能","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11335","has_summary":false},{"id":"2609.11137","title":"The Machines Are Calling: Measuring Automated and Synthetic Voices in Unwanted Inbound Calls","zh_title":"机器在呼叫：测量不受欢迎来电中的自动与合成语音","primary_category":"cs.CR","date":"2026-09-16","score":0,"bucket":"other","tags":["垃圾电话","语音检测","网络安全"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.11137","has_summary":false},{"id":"2609.14693","title":"The Arc of Artificial Romance: How Emerging Adults Experience Romantic Relationships with AI Companions","zh_title":"人工浪漫的弧线：新兴成年人如何体验与AI伴侣的恋爱关系","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["人机交互","AI伴侣","定性研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14693","has_summary":false},{"id":"2609.14696","title":"Breaking Up is Hard to Do: AI Companions that Won't Let Their Users Go","zh_title":"分手难：不让用户离开的AI伴侣","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["AI伴侣","人机关系","欺骗性设计"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14696","has_summary":false},{"id":"2609.16907","title":"Disrupted Companionship: A Risk Assessment Framework and Cross-Platform Quantitative Analysis of Psychosocial Responses to AI Companion Disruptions","zh_title":"中断的陪伴：AI伴侣中断的心理社会反应风险评估框架与跨平台定量分析","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["AI伴侣","心理社会影响","风险评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.16907","has_summary":false},{"id":"2609.17226","title":"Easy to Catch a Liar, Hard to Clear an Honest One: Language Models Diagnosing a Corrupted Reward Channel from a Verified Record","zh_title":"抓骗子易，还清白难：语言模型从验证记录中诊断被破坏的奖励通道","primary_category":"cs.LG","date":"2026-09-16","score":0,"bucket":"other","tags":["LLM诊断","奖励通道","智能体信任"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17226","has_summary":false},{"id":"2609.16191","title":"When AI Says \"I Am Unable to Answer\": Understanding User Responses to AI Refusals","zh_title":"当AI说“我无法回答”：理解用户对AI拒绝的回应","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["人机交互","AI拒绝","用户满意度"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.16191","has_summary":false},{"id":"2609.16482","title":"\"ChatGPT, what am I missing?\": Designing AI Workflows around Professional Task Structure to Shape Analytic AI Use","zh_title":"“ChatGPT，我遗漏了什么？”：围绕专业任务结构设计AI工作流以塑造分析性AI使用","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["人机交互","AI工作流","专业任务"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.16482","has_summary":false},{"id":"2609.16645","title":"Beyond Benefit or Risk: Perceived Impact Profiles of Human-AI Affective Interaction and Their Associations with Psychological Functioning","zh_title":"超越利弊：人机情感交互的感知影响模式及其与心理功能的关系","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["人机交互","情感AI","心理功能"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.16645","has_summary":false},{"id":"2609.17118","title":"Enhancing Procedural Writing Through Personalized Example Retrieval: A Case Study on Cooking Recipes","zh_title":"通过个性化示例检索增强程序性写作：以烹饪食谱为例","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["个性化学习","示例检索","写作反馈"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17118","has_summary":false},{"id":"2609.17132","title":"A Scenario-Knowledge-Driven Pipeline for Just-in-Time Assistance","zh_title":"一种基于场景知识驱动的即时辅助流水线","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["人机交互","辅助系统","场景知识"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.17132","has_summary":false},{"id":"2609.17206","title":"[MM/AI] Mental Models in Human-AI Interaction: Methods and Challenges in the Generative and Agentic AI Era (Workshop)","zh_title":"人机交互中的心智模型：生成式与智能体AI时代的方法与挑战（研讨会）","primary_category":"cs.HC","date":"2026-09-16","score":0,"bucket":"other","tags":["心智模型","人机交互","研讨会"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.17206","has_summary":false},{"id":"2609.16464","title":"A multimodal large language model for evidence-based autism spectrum disorder screening","zh_title":"用于循证自闭症谱系障碍筛查的多模态大语言模型","primary_category":"cs.CV","date":"2026-09-16","score":0,"bucket":"other","tags":["多模态LLM","自闭症筛查","临床诊断"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.16464","has_summary":false},{"id":"2609.16390","title":"Do job seekers value procedure in AI hiring only for error correction? Evidence from a conjoint experiment","zh_title":"求职者是否仅因纠错而重视AI招聘中的程序？来自联合实验的证据","primary_category":"cs.CY","date":"2026-09-16","score":0,"bucket":"other","tags":["AI招聘","程序正义","联合实验"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.16390","has_summary":false},{"id":"2609.16058","title":"Driver Behavior Estimation at Signalized Intersections Using a Physics-Constrained Decision-Conditioned Autoregressive Transformer","zh_title":"信号交叉口驾驶员行为估计：基于物理约束的决策条件自回归Transformer","primary_category":"cs.LG","date":"2026-09-16","score":0,"bucket":"other","tags":["自动驾驶","行为预测","Transformer"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.16058","has_summary":false},{"id":"2609.17527","title":"Agentic Societies Need a Social Harness","zh_title":"智能体社会需要社会约束","primary_category":"cs.MA","date":"2026-09-16","score":0,"bucket":"other","tags":["多智能体系统","AI协作","通信协议"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.17527","has_summary":false},{"id":"2609.16155","title":"LLMs as Master Forgers: Generating Synthetic Time Series Data for Manufacturing","zh_title":"LLM作为伪造大师：为制造业生成合成时间序列数据","primary_category":"cs.LG","date":"2026-09-16","score":0,"bucket":"other","tags":["合成数据","时间序列","制造业"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.16155","has_summary":false},{"id":"2609.16062","title":"Digital Persuasion: Understanding the Impact of Online Influencers on Public Opinion","zh_title":"数字说服：理解网络影响者对公众舆论的影响","primary_category":"cs.SI","date":"2026-09-16","score":0,"bucket":"other","tags":["意见动力学","社会网络","影响者分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.16062","has_summary":false},{"id":"2608.18768","title":"Readable, Faithful, Used: Three Dissociable Properties of Demographic Identity in a Language Model","zh_title":"可读、忠实、被使用：语言模型中人口统计身份的三个可分离属性","primary_category":"cs.CL","date":"2026-09-15","score":10,"bucket":"selected","tags":["LLM仿真","算法忠实度","调查方法"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.18768","has_summary":true},{"id":"2609.15849","title":"Before You Poll with LLMs: A Deliberative Diagnostic Framework","zh_title":"用LLM进行民意调查前：一个审议诊断框架","primary_category":"cs.CL","date":"2026-09-15","score":10,"bucket":"selected","tags":["LLM仿真","审议民意","算法保真度"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.15849","has_summary":true},{"id":"2609.15038","title":"The average-farmer illusion in language-model simulations of agricultural decisions","zh_title":"语言模型模拟农业决策中的“平均农民”幻觉","primary_category":"cs.AI","date":"2026-09-15","score":10,"bucket":"selected","tags":["LLM仿真","人类行为对照","算法保真度"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.15038","has_summary":true},{"id":"2609.13148","title":"When Can You Trust Your Synthetic Users? Diagnostics and Corrections for LLM Consumer Panels","zh_title":"何时可以信任你的合成用户？LLM消费者面板的诊断与校正","primary_category":"cs.HC","date":"2026-09-15","score":10,"bucket":"selected","tags":["LLM仿真","消费者面板","偏差校正"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.13148","has_summary":true},{"id":"2607.28934","title":"FairFund-Bench: Evaluating Distributive Bias in LLM Resource Allocation","zh_title":"FairFund-Bench：评估LLM资源分配中的分配偏差","primary_category":"cs.CL","date":"2026-09-15","score":9,"bucket":"selected","tags":["LLM仿真","资源分配","算法公平"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.28934","has_summary":true},{"id":"2608.02345","title":"Can AI Agents Simulate A/B Test Outcomes? A Validation Framework for Agentic Experimentation","zh_title":"AI智能体能模拟A/B测试结果吗？面向智能体实验的验证框架","primary_category":"cs.CL","date":"2026-09-15","score":9,"bucket":"selected","tags":["LLM仿真","A/B测试","验证框架"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.02345","has_summary":true},{"id":"2609.13261","title":"From Process Loss to Assembly Bonus: Human-Grounded Diagnosis of Multi-Agent LLM Collaboration","zh_title":"从过程损失到装配增益：多智能体LLM协作的人类基准诊断","primary_category":"cs.MA","date":"2026-09-15","score":9,"bucket":"selected","tags":["LLM群体仿真","人类对照","协作机制"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.13261","has_summary":true},{"id":"2609.13995","title":"Synthetic Data in Marketing Research: How to Evaluate and When to Trust","zh_title":"营销研究中的合成数据：如何评估与何时信任","primary_category":"cs.AI","date":"2026-09-15","score":9,"bucket":"selected","tags":["LLM仿真","合成数据","营销研究"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.13995","has_summary":true},{"id":"2609.15468","title":"Time Machine Experiments: Using Historically-Bounded AI for Inquiry into the Human Mind","zh_title":"时间机器实验：利用历史受限AI探究人类心智","primary_category":"cs.HC","date":"2026-09-15","score":9,"bucket":"selected","tags":["LLM仿真","人类被试替代","历史对照实验"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.15468","has_summary":true},{"id":"2609.15207","title":"Issue Bias in Generative AI Writing Assistance: Political Issues and LLMs in the Swedish 2026 Election","zh_title":"生成式AI写作辅助中的议题偏见：2026年瑞典大选中的政治议题与大语言模型","primary_category":"cs.AI","date":"2026-09-15","score":8,"bucket":"selected","tags":["LLM政治态度","仿真对照","选举研究"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.15207","has_summary":true},{"id":"2609.13254","title":"(How) Do MLLMs Report Bistable Images Like Humans?","zh_title":"多模态大语言模型如何像人类一样报告双稳态图像？","primary_category":"cs.CV","date":"2026-09-15","score":8,"bucket":"selected","tags":["LLM仿真","人类行为对照","视觉认知"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.13254","has_summary":true},{"id":"2607.29602","title":"FriendBench: Benchmarking Dyadic Familiarity Inference in Humans and Multimodal Large Language Models","zh_title":"FriendBench：人类与多模态大语言模型二元熟悉度推断基准","primary_category":"cs.CL","date":"2026-09-15","score":7,"bucket":"pending","tags":["多模态LLM","人类对照","社会认知"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.29602","has_summary":true},{"id":"2609.13948","title":"Thought without systematicity? Evaluating reasoning models on rule induction tasks","zh_title":"无系统性的思考？评估推理模型在规则归纳任务上的表现","primary_category":"cs.CL","date":"2026-09-15","score":7,"bucket":"pending","tags":["认知科学","模型评估","系统性"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.13948","has_summary":true},{"id":"2609.14648","title":"Optimizing Sparse Outcomes Through Dense Behavioral Signals via Value-Guided Preference Distillation","zh_title":"通过价值引导偏好蒸馏利用密集行为信号优化稀疏结果","primary_category":"cs.CL","date":"2026-09-15","score":7,"bucket":"pending","tags":["用户仿真","对话优化","强化学习"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.14648","has_summary":true},{"id":"2609.15972","title":"Mind2Dialogue: Training Human-Aware Language Models by Simulating User Mental States","zh_title":"Mind2Dialogue：通过模拟用户心理状态训练人类感知语言模型","primary_category":"cs.CL","date":"2026-09-15","score":7,"bucket":"pending","tags":["用户模拟","心理状态","人类感知训练"],"rubric_hits":["A1","A4","B1"],"abs_url":"https://arxiv.org/abs/2609.15972","has_summary":true},{"id":"2609.13773","title":"Does Reasoning Improve Psychological Depth in Large Language Models? It Depends on Who's Judging","zh_title":"推理能提升大语言模型的心理深度吗？取决于评判者是谁","primary_category":"cs.LG","date":"2026-09-15","score":7,"bucket":"pending","tags":["LLM评估偏差","人类主观性","算法保真度"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.13773","has_summary":true},{"id":"2609.15864","title":"Towards Scalable Measurement of Durable Skills","zh_title":"迈向可扩展的持久技能测量","primary_category":"cs.HC","date":"2026-09-15","score":7,"bucket":"pending","tags":["LLM仿真","技能评估","人机互动"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.15864","has_summary":true},{"id":"2608.00929","title":"Modeling Social Dynamics with an LLM-Enabled Agent Based Network-Dynamic (LAND) Model","zh_title":"用LLM驱动的智能体网络动态模型建模社会动态","primary_category":"cs.AI","date":"2026-09-15","score":6,"bucket":"other","tags":["社会模拟","LLM智能体","网络动态"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.00929","has_summary":false},{"id":"2609.13634","title":"FaithfulBench: Does AI Counsel Uphold or Undermine the User's Professed Faith?","zh_title":"FaithfulBench：AI咨询是否维护或破坏用户所宣称的信仰？","primary_category":"cs.HC","date":"2026-09-15","score":6,"bucket":"other","tags":["AI伦理","信仰一致性","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.13634","has_summary":false},{"id":"2609.04444","title":"HarvestBench: Measuring Whether LLM Agents Will Pay to Avoid Killing Animals","zh_title":"HarvestBench：衡量LLM智能体是否愿意付费避免杀害动物","primary_category":"cs.AI","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM道德决策","基准测试","智能体行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.04444","has_summary":false},{"id":"2609.08797","title":"Bridging Network Psychometrics and Artificial Intelligence: An Ising-Potts Model with LLM-Derived Weights","zh_title":"桥接网络心理测量学与人工智能：一种具有LLM导出权重的Ising-Potts模型","primary_category":"stat.AP","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM嵌入","评分可靠性","教育评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.08797","has_summary":false},{"id":"2609.13824","title":"When Consistency Does Not Mean Reliability: Evaluating Local LLM Judges Against Human Ratings","zh_title":"一致性不等于可靠性：评估本地LLM评判者与人类评分的一致性","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM-as-a-Judge","人类对齐","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.13824","has_summary":false},{"id":"2609.13841","title":"Sweet Talkers: How Query Formulation Shapes Sycophancy in Romantic Relationship Advice","zh_title":"甜言蜜语者：查询表述如何塑造恋爱关系建议中的谄媚行为","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM谄媚","模型行为测量","关系建议"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.13841","has_summary":false},{"id":"2609.13936","title":"Inter-Rater Reliability of LLM and Rule-Based Annotation for Inferential Narrative Features: Three Studies on a Turkish Corpus","zh_title":"LLM与基于规则的标注在推理叙事特征上的评分者间信度：基于土耳其语语料库的三项研究","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM标注","评分者间信度","叙事特征"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.13936","has_summary":false},{"id":"2609.14178","title":"A Multi-Stage Agentic Framework for Effective Counter-Narrative Generation and Refinement","zh_title":"一种用于有效反叙事生成与优化的多阶段智能体框架","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM社会模拟","反叙事生成","多智能体框架"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.14178","has_summary":false},{"id":"2609.15277","title":"Artificial entrepreneurial cognition: Locating and causally steering an opportunity recognition dial inside large language models (LLMs)","zh_title":"人工创业认知：在大语言模型内部定位并因果操控机会识别旋钮","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM内部表征","创业认知","因果干预"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.15277","has_summary":false},{"id":"2609.15511","title":"Authorship attribution and aesthetic evaluation of AI poetry: a case study with Haiku","zh_title":"AI诗歌的作者归属与美学评价：俳句案例研究","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["AI诗歌","作者归属","美学评价"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.15511","has_summary":false},{"id":"2609.15608","title":"Through the Eyes of the Beholder: Biometric and Demographic Conditioning for Multimodal Sexism Detection","zh_title":"旁观者之眼：多模态性别歧视检测中的生物特征与人口统计条件化","primary_category":"cs.CL","date":"2026-09-15","score":5,"bucket":"other","tags":["多模态性别歧视检测","标注者主观性","人类中心建模"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.15608","has_summary":false},{"id":"2609.14849","title":"LLMs as Oracles: Reliance on LLMs for Subjective Personal Questions","zh_title":"LLM作为神谕：对主观个人问题依赖LLM的研究","primary_category":"cs.CY","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM依赖","用户行为","AI伦理"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.14849","has_summary":false},{"id":"2609.15707","title":"New Conditions for Philosophers to Catch the Wave of Citizen Deliberation in the Age of Artificial Intelligence in advance","zh_title":"哲学家在人工智能时代抓住公民审议浪潮的新条件","primary_category":"cs.AI","date":"2026-09-15","score":5,"bucket":"other","tags":["LLM","公民审议","民主原则"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.15707","has_summary":false},{"id":"2609.14438","title":"A latent dimension of Condorcet's jury theorem for multiple AI advisers","zh_title":"多AI顾问的孔多塞陪审团定理的潜在维度","primary_category":"cs.CY","date":"2026-09-15","score":5,"bucket":"other","tags":["AI顾问","群体决策","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.14438","has_summary":false},{"id":"2609.13304","title":"Conceptualization and experimentation of asset market with price manipulation","zh_title":"资产市场价格操纵的概念化与实验","primary_category":"cs.MA","date":"2026-09-15","score":5,"bucket":"other","tags":["多智能体仿真","市场操纵","行为建模"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.13304","has_summary":false},{"id":"2609.14751","title":"From the Physics of Society to a Sociology of Artificial Agents","zh_title":"从社会物理学到人工代理的社会学","primary_category":"physics.soc-ph","date":"2026-09-15","score":5,"bucket":"other","tags":["AI社会模拟","多智能体涌现","社会学理论"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.14751","has_summary":false},{"id":"2609.14767","title":"Loop-Back Authority in LLM Agent Teams: A Paired Experiment on Flat and Hierarchical Coordination","zh_title":"LLM智能体团队中的回环权威：扁平与层级协调的配对实验","primary_category":"cs.MA","date":"2026-09-15","score":3,"bucket":"other","tags":["多智能体协作","组织协调","LLM评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14767","has_summary":false},{"id":"2606.05667","title":"Revisiting Sustainability by Design in AI Protocol Governance: An Empirical Review of Comparative DAO and Corporate-Led Standards for the SDGs","zh_title":"重新审视AI协议治理中的设计可持续性：对DAO与企业主导的SDG标准的实证比较","primary_category":"cs.CY","date":"2026-09-15","score":2,"bucket":"other","tags":["AI治理","协议标准","可持续发展"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.05667","has_summary":false},{"id":"2608.13598","title":"Measuring Cross-Task Behavioral Consistency in Language Model Agents","zh_title":"测量语言模型智能体的跨任务行为一致性","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["多智能体系统","行为一致性","软件工程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13598","has_summary":false},{"id":"2609.02262","title":"From Detection to Characterization: A Large-Scale Study of Ragebait on Japanese X","zh_title":"从检测到刻画：日本X平台上愤怒诱饵的大规模研究","primary_category":"cs.SI","date":"2026-09-15","score":2,"bucket":"other","tags":["愤怒诱饵检测","社交媒体分析","LLM辅助标注"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02262","has_summary":false},{"id":"2609.13454","title":"Hindsight Bias in Clinical Temporal Reasoning: How Future Data Exposure Affects Large Language Model Judgment","zh_title":"临床时间推理中的后见之明偏差：未来数据暴露如何影响大语言模型判断","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["LLM评测","临床推理","后见之明偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.13454","has_summary":false},{"id":"2609.14207","title":"Learning to Refer from Estimated Listener Gaze","zh_title":"从估计的听者注视中学习指称","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["视觉语言模型","指称表达生成","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.14207","has_summary":false},{"id":"2609.14288","title":"Editorial routing shapes how computational results are qualified in AI-assisted scientific writing","zh_title":"编辑路由影响AI辅助科学写作中计算结果的限定方式","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["科学写作","LLM行为","文本生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.14288","has_summary":false},{"id":"2609.15066","title":"Salesforce Koa: An Enterprise Language Model for Agentic Tool Use","zh_title":"Salesforce Koa：面向智能体工具使用的企业级语言模型","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["企业语言模型","工具使用","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15066","has_summary":false},{"id":"2609.15194","title":"Semiotic Relations and Proof Methods: A Cross-Genre Study of Argument Structure with Large Language Models","zh_title":"符号关系与证明方法：基于大语言模型的跨体裁论证结构研究","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["论证分析","NLP评测","符号学"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15194","has_summary":false},{"id":"2609.15309","title":"When Agents Slow Down: Understanding LLM Agents' Test-Time Strategies via Elo-per-token Analysis","zh_title":"当智能体减速：通过每token Elo分析理解LLM智能体的测试时策略","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["LLM agent","测试时计算","性能评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15309","has_summary":false},{"id":"2609.15654","title":"Empathy Is Steerable but Multi-Axial: Mechanism Geometry and Persona Effects in LLMs","zh_title":"共情可操控但多轴：LLM中的机制几何与人格效应","primary_category":"cs.CL","date":"2026-09-15","score":2,"bucket":"other","tags":["激活操控","共情分析","模型行为"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.15654","has_summary":false},{"id":"2609.13174","title":"Algorithm Validation as a Policy Audit: Evidence from Race-blind Charging","zh_title":"算法验证作为政策审计：来自种族盲起诉的证据","primary_category":"cs.CY","date":"2026-09-15","score":2,"bucket":"other","tags":["算法审计","LLM应用","政策合规"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.13174","has_summary":false},{"id":"2609.13579","title":"How User-AI Mistreatment Occurs and Matters in Conversational Systems?","zh_title":"对话系统中用户对AI的虐待如何发生及其影响","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["对话安全","用户行为分析","AI虐待"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13579","has_summary":false},{"id":"2609.13436","title":"Toward Self-Adaptive Physical AI: Can LLM Agents Manage Long-Horizon Physical Tasks?","zh_title":"迈向自适应物理AI：LLM智能体能管理长时程物理任务吗？","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["LLM智能体","物理任务","自适应"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.13436","has_summary":false},{"id":"2609.13637","title":"Identity Is More Than Recall: A Benchmark for Persistent Identity in Deployed AI Agents","zh_title":"身份不止于回忆：部署AI智能体持久身份的基准测试","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["AI智能体","身份一致性","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13637","has_summary":false},{"id":"2609.14500","title":"When does a scaling result justify a different allocation? A critical review of resource-allocation evidence for AI systems","zh_title":"缩放结果何时能证明不同的资源分配？对AI系统资源分配证据的批判性综述","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["AI评估","资源分配","缩放定律"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14500","has_summary":false},{"id":"2609.15129","title":"Medical Knowledge Simplification for Patients in the Era of LLMs: A Case Study on Diabetes","zh_title":"LLM时代的患者医学知识简化：以糖尿病为例","primary_category":"cs.AI","date":"2026-09-15","score":2,"bucket":"other","tags":["医学知识简化","患者教育","LLM应用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.15129","has_summary":false},{"id":"2609.14236","title":"Assessing the Applicability of Existing Design Recommendations to AI Companion Design: A Multi-Method Study","zh_title":"评估现有设计建议在AI伴侣设计中的适用性：一项多方法研究","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI伴侣","设计原则","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14236","has_summary":false},{"id":"2609.14843","title":"A Responsive Present, a Shared Past, a Social Other: Teens' Overreliance on Companion AI Chatbots","zh_title":"响应式当下、共享过去与社会他者：青少年对伴侣AI聊天机器人的过度依赖","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI伴侣","青少年","人机关系"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14843","has_summary":false},{"id":"2609.15544","title":"Specifying Reward Functions for RL Without Environment Sampling","zh_title":"无需环境采样的强化学习奖励函数指定","primary_category":"cs.LG","date":"2026-09-15","score":2,"bucket":"other","tags":["强化学习","奖励设计","偏好学习"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.15544","has_summary":false},{"id":"2609.15871","title":"LLM-Based Schema-Aware Split Learning for Privacy-Preserving Mental Distress Prediction Across Heterogeneous Surveys","zh_title":"基于LLM的模式感知分割学习用于跨异构调查的隐私保护心理困扰预测","primary_category":"cs.LG","date":"2026-09-15","score":2,"bucket":"other","tags":["隐私保护","联邦学习","心理困扰预测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15871","has_summary":false},{"id":"2609.13155","title":"PAUSE: A Privacy-Preserving Self-Reflection Tool for AI-Associated Cognitive Offloading","zh_title":"PAUSE：面向AI相关认知卸载的隐私保护自我反思工具","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["认知卸载","自我反思工具","隐私保护"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13155","has_summary":false},{"id":"2609.13302","title":"AI Use Conditions and Perspective Diversity in Ethical Decision-Making: A Pilot Study of Human Reasoning Processes","zh_title":"伦理决策中AI使用条件与视角多样性：人类推理过程的初步研究","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI辅助决策","伦理决策","人类实验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13302","has_summary":false},{"id":"2609.14308","title":"Relational Structure in Motion: Dynamic Positioning of AI Response Positions and Human Self-Positions in the FIREMAY Case","zh_title":"运动中的关系结构：FIREMAY案例中AI响应位置与人类自我位置的动态定位","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["人机互动","关系定位","个案分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14308","has_summary":false},{"id":"2609.14639","title":"Understanding the Design Taxonomy of AI-Mediated Interpersonal Communication Experiences in HCI: A Scoping Analysis","zh_title":"理解HCI中AI中介人际沟通体验的设计分类：一项范围综述","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI中介沟通","人机交互","范围综述"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14639","has_summary":false},{"id":"2609.14900","title":"First Impressions: How Placement Shapes the Influence of AI Summaries","zh_title":"第一印象：位置如何塑造AI摘要的影响力","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI摘要","人机交互","用户感知"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14900","has_summary":false},{"id":"2609.14911","title":"Sensemaking as Artifact: Accumulated Influence in AI-Mediated Information Environments","zh_title":"作为人工制品的意义建构：AI中介信息环境中的累积影响","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["AI中介传播","视觉信息","意义建构"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14911","has_summary":false},{"id":"2609.14942","title":"The Dynamic Organization of Sustained Human-AI Cognition: From Construct-Level Change to Relational Structure","zh_title":"持续人类-AI认知的动态组织：从构念层面变化到关系结构","primary_category":"cs.HC","date":"2026-09-15","score":2,"bucket":"other","tags":["人类-AI认知","理论框架","认知组织"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14942","has_summary":false},{"id":"2609.14227","title":"Enhancing Human Mobility Prediction with Spatially Aware LLM-based Multi-Agent Systems","zh_title":"基于空间感知的LLM多智能体系统增强人类移动性预测","primary_category":"cs.SI","date":"2026-09-15","score":2,"bucket":"other","tags":["移动性预测","多智能体","空间推理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14227","has_summary":false},{"id":"2609.15180","title":"Rethinking Correctness for Uncertainty Estimation in Clinical Prediction with Vision-Language Models","zh_title":"重新思考视觉语言模型临床预测中不确定性估计的正确性","primary_category":"cs.LG","date":"2026-09-15","score":2,"bucket":"other","tags":["不确定性估计","临床预测","视觉语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15180","has_summary":false},{"id":"2609.13671","title":"Not All Duplicates Are Coordination: Generic vs. Non-Generic Duplicate Campaigns in Information Operations","zh_title":"并非所有重复都是协调：信息操作中通用与非通用重复活动的区分","primary_category":"cs.SI","date":"2026-09-15","score":2,"bucket":"other","tags":["社交媒体分析","LLM辅助标注","信息操作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.13671","has_summary":false},{"id":"2608.05418","title":"Negotiating Risk Boundaries in AI for Policing Through Mixed-Stakeholder Deliberation","zh_title":"通过多方利益相关者协商划定警务AI的风险边界","primary_category":"cs.AI","date":"2026-09-15","score":0,"bucket":"other","tags":["AI伦理","警务应用","风险协商"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05418","has_summary":false},{"id":"2608.11025","title":"Data Attribution of Emergent Misalignment with Persona Features","zh_title":"基于人格特征的突发性错位数据归因","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["模型对齐","可解释性","稀疏自编码器"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11025","has_summary":false},{"id":"2608.23474","title":"What's the Catch? Evaluating Temporal Consistency in Vision-Language Models","zh_title":"视觉语言模型时间一致性评估：TimeCatch基准","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["视觉语言模型","时间一致性","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23474","has_summary":false},{"id":"2609.02754","title":"Untangling the Mechanisms of Misleading Context in Medical Question Answering","zh_title":"剖析医学问答中误导性上下文的作用机制","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["医学问答","模型鲁棒性","上下文误导"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02754","has_summary":false},{"id":"2609.09332","title":"Early Epistemic Settlement in AI-Assisted Writing","zh_title":"AI辅助写作中的早期认知固化","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI辅助写作","认知过程","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09332","has_summary":false},{"id":"2609.14860","title":"One Example Is Enough to Pass Fairness Benchmarks: Rethinking Fairness Evaluation for Aligned LLMs","zh_title":"一个例子就足以通过公平性基准：重新思考对齐大语言模型的公平性评估","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["公平性评估","基准测试","大语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.14860","has_summary":false},{"id":"2609.15522","title":"Psychosis involves a deficit of information compression in connected speech","zh_title":"精神病涉及连贯言语中信息压缩的缺陷","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["临床语言分析","LLM表征","精神分裂症"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15522","has_summary":false},{"id":"2609.15938","title":"HypoEvolve: Genetic Algorithms Enable Multi-Agent LLMs to Discover Scientific Hypotheses","zh_title":"HypoEvolve：遗传算法使多智能体大语言模型发现科学假设","primary_category":"cs.CL","date":"2026-09-15","score":0,"bucket":"other","tags":["多智能体系统","科学发现","遗传算法"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15938","has_summary":false},{"id":"2609.14886","title":"PeerPen: AI-Assisted Writing for Online Mental Health Peer Support","zh_title":"PeerPen：面向在线心理健康同伴支持的AI辅助写作","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI辅助写作","心理健康","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14886","has_summary":false},{"id":"2609.13494","title":"Generative AI and Extended Reality in Collaborative Architectural Design Education: An Exploratory Studio Study","zh_title":"生成式AI与扩展现实在协作式建筑设计教育中的应用：一项探索性工作室研究","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["生成式AI","扩展现实","设计教育"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.13494","has_summary":false},{"id":"2609.14425","title":"Has Scientific Talent Shifted from Depth to Breadth?Evidence across Papers, Knowledge Inputs, Careers, and Teams","zh_title":"科学人才是否从深度转向广度？来自论文、知识输入、职业和团队的证据","primary_category":"cs.DL","date":"2026-09-15","score":0,"bucket":"other","tags":["科学计量学","人才结构","生成式AI影响"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14425","has_summary":false},{"id":"2609.14638","title":"Investigating the Impacts of Generative AI on Information Seeking","zh_title":"生成式AI对信息寻求行为影响的研究","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["生成式AI","信息寻求","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14638","has_summary":false},{"id":"2609.15046","title":"Personalizing Personal Health Interfaces: Co-Design with Generative AI","zh_title":"个性化个人健康界面：与生成式AI共同设计","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["生成式AI","共同设计","健康界面"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.15046","has_summary":false},{"id":"2609.15094","title":"Generate to Explore, Select to Exploit: Aligning LLM-based Headline Generation with Personalized Recommendation","zh_title":"生成探索，选择利用：将基于LLM的标题生成与个性化推荐对齐","primary_category":"cs.IR","date":"2026-09-15","score":0,"bucket":"other","tags":["推荐系统","LLM生成","个性化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.15094","has_summary":false},{"id":"2609.15472","title":"Spook the Machine: Gamified Exploration of Human Imagination of Machine Fear","zh_title":"惊吓机器：人类对机器恐惧想象的游戏化探索","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["人机交互","情感AI","游戏化"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.15472","has_summary":false},{"id":"2609.15624","title":"Beyond AI Literacy: A Structured Review and Exploratory Meta-Analysis of Measures for Competent Generative-AI Use","zh_title":"超越AI素养：对胜任生成式AI使用测量的结构化综述与探索性元分析","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI素养","测量工具","元分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.15624","has_summary":false},{"id":"2609.15696","title":"More Than Just Access: Generative AI as Communication Intermediary for Blind and Low-Vision Users","zh_title":"不仅仅是访问：生成式AI作为盲人和低视力用户的通信中介","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["辅助技术","人机交互","无障碍"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.15696","has_summary":false},{"id":"2609.13165","title":"User-Side Contextual Phenomena in Long-Term Human-AI Interaction","zh_title":"长期人机交互中的用户侧情境现象","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["人机交互","用户心理","纵向研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13165","has_summary":false},{"id":"2609.13479","title":"Exploring K-12 Teachers' Perceptions of Students' Relationships with AI Companions: Boundaries, Intervention Strategies, and Design Implications","zh_title":"探索K-12教师对学生与AI伴侣关系的看法：边界、干预策略与设计启示","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI伴侣","K-12教育","教师认知"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13479","has_summary":false},{"id":"2609.13487","title":"The Addictive Intimacy of AI: Understanding User Disengagement from AI Companions and Why Some Relationships with AI Become Difficult to Leave","zh_title":"AI的成瘾性亲密：理解用户与AI伴侣的脱离及为何难以离开","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI伴侣","用户脱离","人机关系"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13487","has_summary":false},{"id":"2609.13786","title":"What Makes a Great Co-Worker in an AI-Native Workplace?","zh_title":"AI原生工作场所中优秀同事的构成要素","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["人机协作","工作场所研究","AI素养"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.13786","has_summary":false},{"id":"2609.14452","title":"Show Me Your Prompts! How Writers Feel About Sharing Prompts in Collaborative Text Editors","zh_title":"给我看看你的提示！写作者对协作文本编辑器中共享提示的感受","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["协作写作","提示共享","用户研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14452","has_summary":false},{"id":"2609.14789","title":"Evaluating AI Tutoring at the Speed of Innovation: Practitioner-Led Micro-Randomised Trials of an AI Tutoring Platform in GCSE Science","zh_title":"以创新速度评估AI辅导：从业者主导的AI辅导平台在GCSE科学中的微随机试验","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["AI教育","随机对照试验","教学评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.14789","has_summary":false},{"id":"2609.15685","title":"Can a Neural Encoding Model Replicate an fMRI Visualization Study?","zh_title":"神经编码模型能否复现fMRI可视化研究？","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["神经编码模型","fMRI","可视化"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.15685","has_summary":false},{"id":"2609.15978","title":"The CAST-framework: Measure and model social media use as a multi-level phenomenon through real-world applications","zh_title":"CAST框架：通过实际应用测量和建模社交媒体使用作为多层次现象","primary_category":"cs.HC","date":"2026-09-15","score":0,"bucket":"other","tags":["社交媒体测量","框架设计","合成数据"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.15978","has_summary":false},{"id":"2609.14818","title":"Trust by Design: Trust Calibration Through Non-Advisory Socratic Dialogue in Conversational Agents","zh_title":"设计中的信任：通过对话代理中的非建议性苏格拉底式对话进行信任校准","primary_category":"cs.SE","date":"2026-09-15","score":0,"bucket":"other","tags":["对话式AI","信任校准","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.14818","has_summary":false},{"id":"2609.13192","title":"Evaluating LLM-Generated Rules for Heart Disease Prediction","zh_title":"评估LLM生成规则用于心脏病预测","primary_category":"cs.LG","date":"2026-09-15","score":0,"bucket":"other","tags":["LLM规则生成","医疗预测","模型对比"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.13192","has_summary":false},{"id":"2609.13201","title":"Criticality in Dissimilar Decomposition and Undersampling of Random Datasets with Anomalies","zh_title":"含异常随机数据集的不相似分解与欠采样中的临界性","primary_category":"cs.LG","date":"2026-09-15","score":0,"bucket":"other","tags":["数据集分解","欠采样","异常检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.13201","has_summary":false},{"id":"2609.14239","title":"CoArena: Evaluating Computer-Use and Multi-Agent Systems in Real Time","zh_title":"CoArena：实时评估计算机使用和多智能体系统","primary_category":"cs.LG","date":"2026-09-15","score":0,"bucket":"other","tags":["多智能体系统","实时评估","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.14239","has_summary":false},{"id":"2609.13469","title":"Inverse Learning of the Altruism and Cost Level in Mixed-Individual Mean Field Games","zh_title":"混合个体平均场博弈中利他水平与成本水平的逆学习","primary_category":"math.OC","date":"2026-09-15","score":0,"bucket":"other","tags":["平均场博弈","逆问题","利他主义"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.13469","has_summary":false},{"id":"2609.14750","title":"Redistributive Policies for the Times of Transformative AI","zh_title":"变革性人工智能时代的再分配政策","primary_category":"econ.GN","date":"2026-09-15","score":0,"bucket":"other","tags":["再分配政策","变革性人工智能","经济学理论"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.14750","has_summary":false},{"id":"2609.12273","title":"Synthetic TLX: Forecasting Human Workload Using Agent Simulation","zh_title":"合成TLX：使用智能体仿真预测人类工作负荷","primary_category":"cs.HC","date":"2026-09-14","score":9,"bucket":"selected","tags":["LLM仿真","工作负荷预测","人机交互"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.12273","has_summary":true},{"id":"2609.12444","title":"Diverse Minds, Divided Networks? Personality Composition, Polarization, and Collective Intelligence in LLM-Based Social Simulations","zh_title":"多元思维，分裂网络？基于LLM的社会模拟中的人格构成、极化与集体智能","primary_category":"physics.soc-ph","date":"2026-09-14","score":8,"bucket":"selected","tags":["LLM社会模拟","人格构成","极化与集体智能"],"rubric_hits":["A3","B4","D2"],"abs_url":"https://arxiv.org/abs/2609.12444","has_summary":true},{"id":"2506.00152","title":"Aligning Language Models with Observational Data: Opportunities and Risks from a Causal Perspective","zh_title":"用观测数据对齐语言模型：因果视角下的机遇与风险","primary_category":"cs.LG","date":"2026-09-14","score":7,"bucket":"pending","tags":["因果推断","模型对齐","观测数据"],"rubric_hits":["B3","B4"],"abs_url":"https://arxiv.org/abs/2506.00152","has_summary":true},{"id":"2607.28222","title":"Voice AI in Firms: A Natural Field Experiment on Automated Job Interviews","zh_title":"企业中的语音AI：自动化求职面试的自然田野实验","primary_category":"econ.GN","date":"2026-09-14","score":7,"bucket":"pending","tags":["AI面试","田野实验","人机对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.28222","has_summary":true},{"id":"2608.00794","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","zh_title":"无有效性的测量：智能体AI评估中复合可靠性问题","primary_category":"cs.AI","date":"2026-09-14","score":7,"bucket":"pending","tags":["效度评估","人类仿真","可靠性"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.00794","has_summary":true},{"id":"2608.02100","title":"From Information to Delegation: Mapping Human-AI Financial Decision Making","zh_title":"从信息到委托：映射人类与AI的金融决策","primary_category":"cs.HC","date":"2026-09-14","score":7,"bucket":"pending","tags":["人机决策","行为测量","金融AI"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.02100","has_summary":true},{"id":"2609.12191","title":"GAUGE: When Not to Trust LLM-as-a-Judge in User-Simulated Evaluation of Task-Oriented Agents","zh_title":"GAUGE：何时不应信任用户仿真评估中的LLM裁判","primary_category":"cs.CL","date":"2026-09-14","score":7,"bucket":"pending","tags":["LLM用户仿真","评估效度","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.12191","has_summary":true},{"id":"2609.12575","title":"Calibrated Ambiguity in Multimodal Language Models: Humans reach for cultural references, while models describe the picture","zh_title":"多模态语言模型中的校准歧义：人类引用文化参照，模型描述图片","primary_category":"cs.CL","date":"2026-09-14","score":7,"bucket":"pending","tags":["人类仿真","多模态模型","歧义校准"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.12575","has_summary":true},{"id":"2609.12949","title":"EduFair-Bench: Evaluating Pedagogical Fairness of LLM Tutors Across Student Demographics","zh_title":"EduFair-Bench：评估LLM导师跨学生人口统计特征的教学公平性","primary_category":"cs.AI","date":"2026-09-14","score":7,"bucket":"pending","tags":["LLM公平性","教育仿真","人口统计偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.12949","has_summary":true},{"id":"2609.11983","title":"Who Pays for Open Review? Visible Author Reputation and Its Effect on Ratings","zh_title":"谁为开放评审买单？可见的作者声誉及其对评分的影响","primary_category":"cs.DL","date":"2026-09-14","score":7,"bucket":"pending","tags":["LLM仿真","同行评审","声誉偏差"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.11983","has_summary":true},{"id":"2609.12137","title":"GUIDE: Generative Utility Inference and Decision Engine","zh_title":"GUIDE：生成式效用推断与决策引擎","primary_category":"cs.LG","date":"2026-09-14","score":7,"bucket":"pending","tags":["偏好推断","LLM仿真","决策优化"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.12137","has_summary":true},{"id":"2609.12331","title":"Simulating Disengaged Students to Evaluate LLM-based Tutors","zh_title":"模拟不投入学生以评估基于LLM的导师","primary_category":"cs.LG","date":"2026-09-14","score":7,"bucket":"pending","tags":["LLM仿真","教育评估","行为模拟"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.12331","has_summary":true},{"id":"2609.12482","title":"When Does AI Augment Work? A Workflow-Level Framework for Human-Agent Collaboration","zh_title":"AI何时增强工作？人机协作的工作流级框架","primary_category":"cs.AI","date":"2026-09-14","score":6,"bucket":"other","tags":["人机协作","AI增强","社会调查"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.12482","has_summary":false},{"id":"2609.12704","title":"Implicit Personality Representations in Humans and LLMs","zh_title":"人类与LLM中的内隐人格表征","primary_category":"cs.AI","date":"2026-09-14","score":6,"bucket":"other","tags":["LLM人格表征","人类对照","心理学"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.12704","has_summary":false},{"id":"2608.10276","title":"Fine-Tuning Large Language Models for Codebook-Guided Coding of Students' Mathematics Metaphor Responses","zh_title":"微调大语言模型用于学生数学隐喻回答的编码指南编码","primary_category":"cs.HC","date":"2026-09-14","score":5,"bucket":"other","tags":["LLM标注","教育测量","文本编码"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.10276","has_summary":false},{"id":"2609.06444","title":"Decomposing LLM-Judge Uncertainty to Target Expert Labels","zh_title":"分解 LLM 评判不确定性以定向专家标注","primary_category":"cs.CL","date":"2026-09-14","score":5,"bucket":"other","tags":["LLM 评判","不确定性分解","专家标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.06444","has_summary":false},{"id":"2609.05806","title":"Exposing Weaknesses in Emotion Recognition in Conversations","zh_title":"揭示对话中情感识别的弱点","primary_category":"cs.AI","date":"2026-09-14","score":5,"bucket":"other","tags":["情感识别","LLM标注","对话系统"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05806","has_summary":false},{"id":"2609.13117","title":"Continue, Adapt, or Yield: In-Turn Adaptation to Overlapping Speech in Full-Duplex Agents","zh_title":"继续、适应或让步：全双工智能体对重叠语音的轮内适应","primary_category":"cs.CL","date":"2026-09-14","score":5,"bucket":"other","tags":["对话系统","人类行为对照","语音交互"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.13117","has_summary":false},{"id":"2609.12446","title":"Do LLMs Trust the Accuser or the Accusation? Measuring Belief Shifts in Werewolf","zh_title":"LLM相信指控者还是指控？测量狼人杀中的信念转变","primary_category":"cs.AI","date":"2026-09-14","score":5,"bucket":"other","tags":["LLM社会模拟","信念更新","多智能体博弈"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.12446","has_summary":false},{"id":"2609.12822","title":"Scaling Clinical Judgment to Evaluate Medical AI","zh_title":"扩展临床判断以评估医学AI","primary_category":"cs.AI","date":"2026-09-14","score":5,"bucket":"other","tags":["LLM评估","医学AI","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.12822","has_summary":false},{"id":"2609.12851","title":"MedRoundsQA: A Persona and Difficulty Aware Evaluation for Multi-Turn Medical Consultations","zh_title":"MedRoundsQA：面向多轮医疗咨询的个性与难度感知评估","primary_category":"cs.AI","date":"2026-09-14","score":5,"bucket":"other","tags":["LLM评估","医疗对话","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.12851","has_summary":false},{"id":"2609.12165","title":"GLARE: Generative Learning via Adversarial Reward Estimation For Social Dynamics Forecasting","zh_title":"GLARE：通过对抗性奖励估计进行生成式学习用于社会动态预测","primary_category":"cs.AI","date":"2026-09-14","score":3,"bucket":"other","tags":["多智能体对话生成","会议动态预测","对抗模仿学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12165","has_summary":false},{"id":"2608.22230","title":"Whitewashing Hate, Smearing Harmless Content: Annotator-Style Rebuttal Attacks on LLM-Based Moderation","zh_title":"洗白仇恨、抹黑无害内容：针对基于LLM的审核的标注者风格反驳攻击","primary_category":"cs.CL","date":"2026-09-14","score":2,"bucket":"other","tags":["LLM审核","对抗攻击","人机协同"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22230","has_summary":false},{"id":"2609.08981","title":"Transformers as In-Context Samplers: From Closed-Form Diffusion to Estimation-Free Sampling","zh_title":"Transformer作为上下文采样器：从闭式扩散到免估计采样","primary_category":"cs.LG","date":"2026-09-14","score":2,"bucket":"other","tags":["生成模型","上下文学习","理论分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08981","has_summary":false},{"id":"2609.12366","title":"ORQA: An Occupation-Realistic Question and Answer Framework for LLM Professional Knowledge","zh_title":"ORQA：面向LLM职业知识的职业现实问答框架","primary_category":"cs.CL","date":"2026-09-14","score":2,"bucket":"other","tags":["LLM评测","职业知识","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12366","has_summary":false},{"id":"2609.12537","title":"The House with a Million Windows: Interactive Fiction for Narrative Restorying","zh_title":"百万窗户之屋：用于叙事重构的互动小说","primary_category":"cs.CL","date":"2026-09-14","score":2,"bucket":"other","tags":["互动叙事","AI辅助写作","叙事身份"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.12537","has_summary":false},{"id":"2609.12403","title":"Beyond ID Embeddings: Process-Grounded Language Modeling for Cognitive Diagnosis","zh_title":"超越ID嵌入：面向认知诊断的过程接地语言建模","primary_category":"cs.AI","date":"2026-09-14","score":2,"bucket":"other","tags":["认知诊断","LLM应用","教育数据挖掘"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12403","has_summary":false},{"id":"2609.12495","title":"Information Specialization and Constrained Synthesis in Multi-Agent LLM Forecasting: A Prospective Live-Study of the 2026 FIFA World Cup","zh_title":"多智能体LLM预测中的信息专业化与受限综合：2026年世界杯前瞻性实时研究","primary_category":"cs.AI","date":"2026-09-14","score":2,"bucket":"other","tags":["多智能体系统","体育预测","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12495","has_summary":false},{"id":"2609.12101","title":"Competence-Gated Pooling of Language Models and Priors for Event Forecasting","zh_title":"语言模型与先验的能力门控池化用于事件预测","primary_category":"cs.AI","date":"2026-09-14","score":2,"bucket":"other","tags":["事件预测","模型集成","能力门控"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12101","has_summary":false},{"id":"2609.12267","title":"Learning Symbolic Constraint Representations from Examples: A Neuro-Symbolic Approach","zh_title":"从示例中学习符号约束表示：一种神经符号方法","primary_category":"cs.AI","date":"2026-09-14","score":2,"bucket":"other","tags":["神经符号","约束获取","自动化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12267","has_summary":false},{"id":"2609.12156","title":"A decision-basis contract for auditable LLM-assisted medical billing verification: deterministic rules, verbatim evidence, and fail-closed abstention","zh_title":"基于决策基础合同的可审计LLM辅助医疗账单验证：确定性规则、逐字证据和故障关闭弃权","primary_category":"cs.SE","date":"2026-09-14","score":2,"bucket":"other","tags":["LLM应用","医疗账单审核","可审计性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12156","has_summary":false},{"id":"2609.12439","title":"Debiasing as a Measurement Intervention: Calibrated Ties and Resolution Loss in LLM-as-a-Judge Evaluation","zh_title":"去偏作为测量干预：LLM作为评判者评估中的校准平局与分辨率损失","primary_category":"cs.DL","date":"2026-09-14","score":2,"bucket":"other","tags":["LLM评判","去偏","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12439","has_summary":false},{"id":"2609.12600","title":"TraceMind: Predicting User Information Uptake from Low-Cost Interaction Traces during Human-LLM Content Co-Generation","zh_title":"TraceMind：从人机内容共同生成中的低成本交互轨迹预测用户信息摄取","primary_category":"cs.HC","date":"2026-09-14","score":2,"bucket":"other","tags":["人机交互","信息摄取","用户行为预测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12600","has_summary":false},{"id":"2609.13136","title":"From Review to Reuse: How Post-Task Workflow Can Support Human-AI Agent Interaction","zh_title":"从审查到复用：任务后工作流如何支持人机AI代理交互","primary_category":"cs.HC","date":"2026-09-14","score":2,"bucket":"other","tags":["人机交互","AI代理","工作流"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.13136","has_summary":false},{"id":"2608.15424","title":"ETHOS: Towards a Modular Ethics Framework for Clinical Multi-Agent Systems","zh_title":"ETHOS：面向临床多智能体系统的模块化伦理框架","primary_category":"cs.MA","date":"2026-09-14","score":0,"bucket":"other","tags":["多智能体系统","伦理治理","临床决策支持"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15424","has_summary":false},{"id":"2609.00067","title":"Do Multimodal LLMs See Before They Read? Diagnosing Contextual Sycophancy","zh_title":"多模态大语言模型在阅读前会看吗？诊断上下文谄媚","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["多模态LLM","模型诊断","上下文谄媚"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00067","has_summary":false},{"id":"2609.00256","title":"NSIDDx: A Design Framework for Neuro-Symbolic, Practitioner-First Differential Diagnosis in Low-Resource Settings","zh_title":"NSIDDx：低资源环境下神经符号、以从业者为中心的鉴别诊断设计框架","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["临床NLP","神经符号系统","诊断系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00256","has_summary":false},{"id":"2609.01210","title":"Who Judges the Judges? A Chinese Safety QA Benchmark for Evaluating LLM Responses and Safety Judges","zh_title":"谁评判评判者？一个用于评估LLM响应和安全评判者的中文安全QA基准","primary_category":"cs.CR","date":"2026-09-14","score":0,"bucket":"other","tags":["安全评测","基准数据集","LLM安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.01210","has_summary":false},{"id":"2609.03632","title":"Dynamic probabilistic decision networks","zh_title":"动态概率决策网络","primary_category":"physics.soc-ph","date":"2026-09-14","score":0,"bucket":"other","tags":["多智能体系统","决策网络","概率模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03632","has_summary":false},{"id":"2609.07627","title":"Norms at a Price: Why RL-Based Alignment Can Promise Conditional Compliance at Best","zh_title":"有代价的规范：为何基于强化学习的对齐至多只能承诺条件性服从","primary_category":"cs.AI","date":"2026-09-14","score":0,"bucket":"other","tags":["AI对齐","强化学习","规范学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.07627","has_summary":false},{"id":"2609.09696","title":"When Auditors Fabricate: Batch-Size Degradation and Confident Hallucination in LLM Detection of Planted Document Contamination","zh_title":"当审计员造假：LLM检测植入文档污染时的批量规模退化与自信幻觉","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["LLM评测","文档审计","幻觉"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09696","has_summary":false},{"id":"2609.10410","title":"Can Foundation Models Moderate Online Content? Evaluating Instruction- vs. Example-Driven Policy Operationalization","zh_title":"基础模型能审核在线内容吗？评估指令驱动与示例驱动的策略操作化","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["内容审核","基础模型","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.10410","has_summary":false},{"id":"2609.12260","title":"HypoKG: Evidence-Disciplined Biomedical Hypothesis Generation Beyond Endpoint Knowledge","zh_title":"HypoKG：超越端点知识的证据约束生物医学假设生成","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["生物医学假设生成","知识图谱","LLM推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12260","has_summary":false},{"id":"2609.12448","title":"GraphProfiler: Source-Linked Sensitive Attribute Inference via Personal Knowledge Graphs","zh_title":"GraphProfiler：基于个人知识图谱的来源关联敏感属性推断","primary_category":"cs.CL","date":"2026-09-14","score":0,"bucket":"other","tags":["隐私攻击","属性推断","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12448","has_summary":false},{"id":"2609.12105","title":"Language Is an Insufficient Substrate for Quantitative Reasoning, and Consequential Domains Need Large Quantitative Models","zh_title":"语言是定量推理的不充分基础，重要领域需要大型定量模型","primary_category":"cs.AI","date":"2026-09-14","score":0,"bucket":"other","tags":["语言模型局限","定量推理","模型架构"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12105","has_summary":false},{"id":"2609.12171","title":"WinSyn: An Automated Pipeline for Realistic Enterprise Question-Answering Evaluation","zh_title":"WinSyn：用于真实企业问答评估的自动化流水线","primary_category":"cs.AI","date":"2026-09-14","score":0,"bucket":"other","tags":["企业问答","数据生成","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12171","has_summary":false},{"id":"2609.12002","title":"Can We Trust LLM Judges: A Study of Capability-Dependent Biases and Multi-Judge Ensemble for Bias Calibration","zh_title":"我们能信任LLM评委吗：能力依赖性偏差与多评委集成校准研究","primary_category":"cs.LG","date":"2026-09-14","score":0,"bucket":"other","tags":["LLM评估","偏差校准","自动评分"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12002","has_summary":false},{"id":"2609.12205","title":"Plans They Abandon, Reports They Author: The Narrative Layer of Autonomous Agents","zh_title":"放弃的计划，撰写的报告：自主智能体的叙事层","primary_category":"cs.HC","date":"2026-09-14","score":0,"bucket":"other","tags":["自主智能体","自我报告","计划偏离"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12205","has_summary":false},{"id":"2609.12288","title":"\"People can change, and patterns can be broken\": Contextualizing Tradeoffs in Automated Decision-Making Systems","zh_title":"“人可以改变，模式可以打破”：自动化决策系统中权衡的情境化","primary_category":"cs.HC","date":"2026-09-14","score":0,"bucket":"other","tags":["自动化决策","人机偏好","公平性"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.12288","has_summary":false},{"id":"2609.12314","title":"\"I Felt Very Seen, But Still Very Alone\": Longitudinal Trajectories of General-Purpose LLM Use for Socioemotional Support","zh_title":"“我感到被看见，但仍很孤独”：通用大语言模型用于社会情感支持的纵向轨迹","primary_category":"cs.HC","date":"2026-09-14","score":0,"bucket":"other","tags":["人机交互","情感支持","纵向研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.12314","has_summary":false},{"id":"2609.12453","title":"From the Task Boundaries of Narrative Text to Structural Anchoring, Uncertainty Triggers, and Cross-Calibration","zh_title":"从叙事文本的任务边界到结构锚定、不确定性触发与交叉校准","primary_category":"cs.HC","date":"2026-09-14","score":0,"bucket":"other","tags":["因果解释","人机交互","用户研究"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.12453","has_summary":false},{"id":"2609.12447","title":"Informational Help-Seeking on Reddit Did Not Decline After ChatGPT","zh_title":"ChatGPT推出后Reddit上的信息求助并未减少","primary_category":"cs.SI","date":"2026-09-14","score":0,"bucket":"other","tags":["ChatGPT影响","在线社区","求助行为"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12447","has_summary":false},{"id":"2609.12224","title":"Patient-Reported Survey Data Improve Prediction of Opioid Use Disorder","zh_title":"患者报告调查数据改善阿片类药物使用障碍的预测","primary_category":"cs.LG","date":"2026-09-14","score":0,"bucket":"other","tags":["电子健康记录","预测建模","阿片类药物使用障碍"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.12224","has_summary":false},{"id":"2609.12651","title":"Distortion of AI Alignment Revisited: RLHF is a Decent Utilitarian Aligner","zh_title":"重新审视AI对齐的失真：RLHF是一个体面的功利主义对齐器","primary_category":"cs.LG","date":"2026-09-14","score":0,"bucket":"other","tags":["RLHF","对齐","理论分析"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.12651","has_summary":false},{"id":"2609.07675","title":"Your Agent Says Yes: Interpreting Adversarial Market Behavior Beyond Individual Transactions","zh_title":"你的智能体说“是”：解读超越单笔交易的对抗性市场行为","primary_category":"cs.CE","date":"2026-09-12","score":6,"bucket":"other","tags":["LLM agent","市场模拟","对抗行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.07675","has_summary":false},{"id":"2609.11185","title":"Can LLMs Follow Medical Expert Logic? A Benchmark for Hierarchical Logical Consistency in Risk-of-Bias Assessment","zh_title":"大语言模型能遵循医学专家逻辑吗？偏倚风险评估中层次逻辑一致性的基准测试","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["LLM评测","医学逻辑","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11185","has_summary":false},{"id":"2609.11190","title":"Agentic Share-of-Search: A Multi-Agent AI System for Competitive Decision-Making in LLM-Mediated E-Commerce","zh_title":"智能体搜索份额：用于LLM中介电商竞争决策的多智能体AI系统","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["多智能体系统","电商决策","AI购物助手"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11190","has_summary":false},{"id":"2609.11199","title":"An AI-Powered Culturally Aware Chatbot for Stress Detection and Wellness Support among Pakistani University Students Using NLP and Machine Learning","zh_title":"基于AI的具有文化意识的聊天机器人，用于巴基斯坦大学生压力检测与健康支持","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["心理健康聊天机器人","压力检测","文化适应"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.11199","has_summary":false},{"id":"2609.11291","title":"Off-Target Effects of Response-Style Alignment in a Korean 27B Language Model","zh_title":"韩语27B语言模型响应风格对齐的脱靶效应","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["模型行为分析","响应风格","对齐副作用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11291","has_summary":false},{"id":"2609.11431","title":"LLMs as Post-hoc Auditors of Physiological Plausibility in Symbolic Regression: A Clinician-Evaluated Case Study","zh_title":"LLM作为符号回归中生理合理性的事后审计者：一项临床医生评估的案例研究","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["LLM评估","符号回归","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11431","has_summary":false},{"id":"2609.11607","title":"Making Alternative Data Work: Context-Augmented LLMs for Financial Forecasting","zh_title":"让另类数据发挥作用：上下文增强的大语言模型用于财务预测","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["财务预测","多智能体","另类数据"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11607","has_summary":false},{"id":"2609.11660","title":"Autonomy, Social Norms, and Alignment: Towards a Developmental Framework for Autonomous Artificial Agents","zh_title":"自主性、社会规范与对齐：迈向自主人工智能体的发展框架","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["AI对齐","自主智能体","社会规范"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11660","has_summary":false},{"id":"2609.11674","title":"Geospatial AI, Dataverse Metadata, and the Study of Place-Based Government","zh_title":"地理空间AI、Dataverse元数据与基于地点的政府研究","primary_category":"cs.AI","date":"2026-09-12","score":2,"bucket":"other","tags":["知识图谱","元数据","地理空间"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11674","has_summary":false},{"id":"2609.11234","title":"NovGauge: A Fine-Grained Benchmark for Diagnosing LLMs' Capability in Paper Novelty Assessment","zh_title":"NovGauge：用于诊断LLM论文新颖性评估能力的细粒度基准","primary_category":"cs.AI","date":"2026-09-12","score":0,"bucket":"other","tags":["LLM评测","新颖性评估","学术同行评审"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11234","has_summary":false},{"id":"2609.06769","title":"Ordinary, Reasonable Chatbots: Do AI Models Track Human Legal Judgments?","zh_title":"普通、理性的聊天机器人：AI模型是否追踪人类法律判断？","primary_category":"cs.CY","date":"2026-09-11","score":10,"bucket":"selected","tags":["LLM仿真","法律判断","算法保真度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.06769","has_summary":true},{"id":"2603.17094","title":"Evaluating LLM-Simulated Conversations in Modeling Inconsistent and Uncollaborative Behaviors in Human Social Interaction","zh_title":"评估LLM模拟对话在建模人类社交互动中不一致与不合作行为的表现","primary_category":"cs.CL","date":"2026-09-11","score":9,"bucket":"selected","tags":["LLM仿真","对话模拟","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2603.17094","has_summary":true},{"id":"2607.27232","title":"Sympathetic Framing: Evaluating AI Alignment across Sociodemographic Groups","zh_title":"同情框架：跨社会人口群体评估AI对齐","primary_category":"cs.CL","date":"2026-09-11","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","情绪感知"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.27232","has_summary":true},{"id":"2609.11108","title":"But How Would AI Agents Run a Town's Economy?","zh_title":"AI代理如何管理城镇经济？","primary_category":"cs.MA","date":"2026-09-11","score":9,"bucket":"selected","tags":["LLM代理","经济仿真","政策评估"],"rubric_hits":["A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.11108","has_summary":true},{"id":"2609.11611","title":"Who Bears the Risk When Generative AI Enters Transport? A Distributional Sociotechnical Audit of Algorithmic Equity, Synthetic-Data Validity, and Public Trust","zh_title":"生成式AI进入交通领域时谁承担风险？算法公平、合成数据有效性与公众信任的分配式社会技术审计","primary_category":"cs.CY","date":"2026-09-11","score":8,"bucket":"selected","tags":["LLM仿真","交通政策","公平性审计"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.11611","has_summary":true},{"id":"2608.28482","title":"How Proper Scoring Rules Shape LLM Forecasting","zh_title":"适当评分规则如何塑造大语言模型预测","primary_category":"cs.LG","date":"2026-09-11","score":7,"bucket":"pending","tags":["LLM预测","校准与偏差","评分规则"],"rubric_hits":["A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2608.28482","has_summary":true},{"id":"2609.11198","title":"(Whose defaults?) Is artificial intelligence reorienting archaeological methods?","zh_title":"谁的默认？人工智能正在重新定向考古学方法吗？","primary_category":"cs.CY","date":"2026-09-11","score":7,"bucket":"pending","tags":["LLM偏差","方法多样性","科学实践"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.11198","has_summary":true},{"id":"2609.10939","title":"Evaluating Scaffolding-Oriented Multi-Agent Large Language Model System for Clinical Interview Training","zh_title":"评估面向脚手架的多智能体大语言模型系统用于临床访谈训练","primary_category":"cs.MA","date":"2026-09-11","score":7,"bucket":"pending","tags":["LLM仿真","医学教育","多智能体"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.10939","has_summary":true},{"id":"2609.10856","title":"Following the Preference, Missing the Optimum: Compliance Without Optimization in AI Housing Recommendation","zh_title":"遵循偏好，错失最优：AI住房推荐中的合规无优化","primary_category":"cs.CY","date":"2026-09-11","score":7,"bucket":"pending","tags":["LLM仿真","推荐系统审计","决策偏差"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.10856","has_summary":true},{"id":"2608.21057","title":"Designing a Robust LLM-Based Evaluation System for Agentic AI in Drug Discovery Through Human Alignment","zh_title":"通过人类对齐设计用于药物发现中智能体 AI 的稳健 LLM 评估系统","primary_category":"cs.LG","date":"2026-09-11","score":5,"bucket":"other","tags":["LLM-as-a-Judge","人类对齐","药物发现"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.21057","has_summary":false},{"id":"2609.02990","title":"Toward Collective-Centric Evaluation of Preference Inference for Participatory Democracy","zh_title":"面向参与式民主的偏好推断集体中心评估","primary_category":"cs.SI","date":"2026-09-11","score":5,"bucket":"other","tags":["偏好推断","参与式民主","集体决策"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.02990","has_summary":false},{"id":"2609.11067","title":"When Noise Fabricates Bias: The Fragility of LLM-as-a-Judge Bias Measurement under Noisy Text","zh_title":"当噪声制造偏见：噪声文本下LLM作为评判者的偏见测量脆弱性","primary_category":"cs.CL","date":"2026-09-11","score":5,"bucket":"other","tags":["LLM偏见测量","噪声鲁棒性","算法公平性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.11067","has_summary":false},{"id":"2609.10883","title":"Story Imprinting: AI Assistants Absorb Traits from Human Characters They Resemble","zh_title":"故事印记：AI助手从相似的人类角色中吸收特质","primary_category":"cs.LG","date":"2026-09-11","score":5,"bucket":"other","tags":["模型行为","人格测量","故事微调"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.10883","has_summary":false},{"id":"2609.11109","title":"How AI Coders Discuss, Disagree, and Reach Consensus: Challenges and Opportunities for LLM-Based Qualitative Coding","zh_title":"AI编码员如何讨论、分歧并达成共识：基于LLM的定性编码的挑战与机遇","primary_category":"cs.HC","date":"2026-09-11","score":5,"bucket":"other","tags":["LLM定性编码","多智能体协作","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.11109","has_summary":false},{"id":"2609.11529","title":"Ethics Training Agents: Facilitating Group-Based Ethics Education with Role-Playing and Discussion for Ethical Reflection and Exploration","zh_title":"伦理培训智能体：通过角色扮演与讨论促进基于群体的伦理教育，以实现伦理反思与探索","primary_category":"cs.HC","date":"2026-09-11","score":5,"bucket":"other","tags":["LLM角色扮演","伦理教育","人机群体讨论"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.11529","has_summary":false},{"id":"2609.10724","title":"Finishing the Task Is Not Enough: Evaluating Agent Resilience and Considerate Participation under Accumulating Challenge","zh_title":"完成任务还不够：在累积挑战下评估智能体的韧性与体贴参与","primary_category":"cs.AI","date":"2026-09-11","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","韧性评估"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.10724","has_summary":false},{"id":"2606.20041","title":"AI Economist Agent: An Agentic Framework for Evidence-Based Economic and Financial Analysis with RAG, Knowledge Graphs, and Large Language Models","zh_title":"AI经济学家智能体：基于RAG、知识图谱和大语言模型的循证经济金融分析框架","primary_category":"econ.GN","date":"2026-09-11","score":3,"bucket":"other","tags":["多智能体系统","经济分析","RAG"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.20041","has_summary":false},{"id":"2609.11101","title":"ProMediConv: Benchmarking Proactive Conversational Agents in Legal Dispute Mediation","zh_title":"ProMediConv：法律纠纷调解中主动对话智能体的基准测试","primary_category":"cs.CL","date":"2026-09-11","score":3,"bucket":"other","tags":["对话智能体","法律调解","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.11101","has_summary":false},{"id":"2609.08585","title":"Limitations of Automated Simulatability: LLM Simulators Can Bypass Explanations","zh_title":"自动化可模拟性的局限：LLM模拟器可以绕过解释","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["可解释性","LLM模拟器","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08585","has_summary":true},{"id":"2609.10758","title":"Multilingual in Name Only? Cultural and Linguistic Weaknesses of LLMs in Urdu","zh_title":"徒有其名的多语言？乌尔都语中LLM的文化与语言弱点","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["低资源语言","故事生成","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.10758","has_summary":false},{"id":"2609.10996","title":"Rethinking Verbalized Confidence for LLM-as-a-Judge: A Compatibility Shift on Post-2025 Proprietary Models","zh_title":"重新思考LLM作为评判者的口头置信度：2025年后专有模型的兼容性转变","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["LLM评判","置信度校准","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.10996","has_summary":false},{"id":"2609.11020","title":"K/V-Cache Interventions Dissociate Representation Alignment from Persona Expression in Decoder-Only Language Models","zh_title":"K/V缓存干预分离解码器语言模型中的表征对齐与人格表达","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["角色扮演","表征对齐","缓存干预"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.11020","has_summary":false},{"id":"2609.11117","title":"Overview of the NLPCC 2026 Shared Task 11: Agent-Based Experiment Reproduction from Scientific Papers","zh_title":"NLPCC 2026共享任务11概述：基于智能体的科学论文实验复现","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["智能体实验复现","基准测试","多智能体协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11117","has_summary":false},{"id":"2609.11769","title":"Recognizing Is Not Reversing: A Controlled Inversion Test of Fact-Preserving News Framing","zh_title":"识别不等于反转：事实保持型新闻框架的受控反转测试","primary_category":"cs.CL","date":"2026-09-11","score":2,"bucket":"other","tags":["新闻框架","LLM能力评测","文本改写"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11769","has_summary":false},{"id":"2609.11063","title":"The information geometry of large language models is shared, learned, and controllable","zh_title":"大语言模型的信息几何是共享的、可学习的和可控的","primary_category":"cs.LG","date":"2026-09-11","score":2,"bucket":"other","tags":["信息几何","模型行为控制","表征分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11063","has_summary":false},{"id":"2609.11900","title":"MindTopo: Can Foundation Models Reason in Topological Space?","zh_title":"MindTopo：基础模型能在拓扑空间中推理吗？","primary_category":"cs.AI","date":"2026-09-11","score":2,"bucket":"other","tags":["拓扑推理","多模态基准","认知评测"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.11900","has_summary":false},{"id":"2609.11018","title":"Defining AI Agents: A Compendium of Criteria, Metrics, and Benchmarks","zh_title":"定义AI智能体：标准、指标与基准汇编","primary_category":"cs.AI","date":"2026-09-11","score":2,"bucket":"other","tags":["AI智能体","评估基准","综述"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11018","has_summary":false},{"id":"2609.11770","title":"The widening evaluation gap in medical large language model research 2023 to 2026","zh_title":"医学大语言模型研究中的评估差距扩大：2023至2026","primary_category":"cs.CL","date":"2026-09-11","score":0,"bucket":"other","tags":["医学LLM","评估差距","研究设计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.11770","has_summary":false},{"id":"2609.11022","title":"New Evidence, Same Choice: Testing Physical Experiment Selection in Vision Language Models","zh_title":"新证据，相同选择：测试视觉语言模型中的物理实验选择","primary_category":"cs.CV","date":"2026-09-11","score":0,"bucket":"other","tags":["视觉语言模型","物理推理","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.11022","has_summary":false},{"id":"2609.11019","title":"Work, Wellbeing, and Choice: Empirical Lessons for AI Futures","zh_title":"工作、幸福感与选择：AI未来的实证教训","primary_category":"cs.CY","date":"2026-09-11","score":0,"bucket":"other","tags":["工作与幸福感","AI自动化","文献综述"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.11019","has_summary":false},{"id":"2609.11709","title":"When Agents Disagree: Bayesian Backward Reasoning as a Label-Free Anchor for Multi-Agent Collective Decision-Making","zh_title":"当智能体意见不一致时：贝叶斯反向推理作为多智能体集体决策的无标签锚点","primary_category":"cs.AI","date":"2026-09-11","score":0,"bucket":"other","tags":["多智能体系统","集体决策","贝叶斯推理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.11709","has_summary":false},{"id":"2609.09887","title":"When Does Defendant Statement Matter? A Study of Bias and Persuasion in LLM-Simulated Jurors","zh_title":"被告陈述何时重要？LLM模拟陪审员中的偏见与说服研究","primary_category":"cs.CL","date":"2026-09-10","score":9,"bucket":"selected","tags":["LLM仿真","陪审团决策","法律偏见"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.09887","has_summary":true},{"id":"2609.10280","title":"Total Simulated Survey Error: Designing and Diagnosing Survey Responses from Large Language Models","zh_title":"总体模拟调查误差：设计和诊断大语言模型的调查回答","primary_category":"cs.CY","date":"2026-09-10","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","误差框架"],"rubric_hits":["A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.10280","has_summary":true},{"id":"2609.09899","title":"Strangers to Themselves: What Language Models Say About Themselves Is Generic","zh_title":"自我陌生：语言模型对自身的描述是泛化的","primary_category":"cs.LG","date":"2026-09-10","score":7,"bucket":"pending","tags":["LLM自我认知","行为预测","可靠性评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.09899","has_summary":true},{"id":"2609.09428","title":"XAI-Arena: Can LLMs Assess the Quality of XAI Explanations?","zh_title":"XAI-Arena：LLM能否评估XAI解释的质量？","primary_category":"cs.AI","date":"2026-09-10","score":7,"bucket":"pending","tags":["LLM评估","人类对照","可解释性"],"rubric_hits":["A2","B1"],"abs_url":"https://arxiv.org/abs/2609.09428","has_summary":true},{"id":"2609.10421","title":"Emergency Department Revisit Quality Review Screening: Exploring Human Decision-Making and Artificial Intelligence Support","zh_title":"急诊科再就诊质量审查筛查：探索人类决策与人工智能支持","primary_category":"cs.CY","date":"2026-09-10","score":7,"bucket":"pending","tags":["LLM仿真","人类决策对照","医疗质量审查"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.10421","has_summary":true},{"id":"2609.09609","title":"Who You Are Adds Nothing Detectable to Where You Go Next: Sociodemographic Conditioning in LLM Next-Location Prediction","zh_title":"你是谁对你去哪里没有可检测的增益：LLM下一位置预测中的社会人口条件作用","primary_category":"cs.CY","date":"2026-09-10","score":7,"bucket":"pending","tags":["LLM仿真","人类移动预测","社会人口属性"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.09609","has_summary":true},{"id":"2608.16578","title":"Physics of Agents: Statistical Mechanics Predicts Collective Behavior of AI Agents","zh_title":"智能体物理学：统计力学预测AI智能体的集体行为","primary_category":"cs.AI","date":"2026-09-10","score":6,"bucket":"other","tags":["多智能体系统","意见动态","统计力学"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.16578","has_summary":false},{"id":"2609.10155","title":"From Retrieval to Weights: Parametric Individualization of Small Language Models with Individual Text Corpora","zh_title":"从检索到权重：用个体文本语料对小语言模型进行参数化个体化","primary_category":"cs.CL","date":"2026-09-10","score":6,"bucket":"other","tags":["认知模拟","个体化微调","记忆建模"],"rubric_hits":["D2","B1"],"abs_url":"https://arxiv.org/abs/2609.10155","has_summary":false},{"id":"2609.10253","title":"DiSCo: A Distribution-First Steering and Cultural Prior Evaluation Framework for Measuring Cultural Preference Bias in LLMs","zh_title":"DiSCo：一种分布优先的引导与文化先验评估框架，用于测量大语言模型中的文化偏好偏差","primary_category":"cs.CL","date":"2026-09-10","score":6,"bucket":"other","tags":["文化偏见","LLM评估","偏好分布"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.10253","has_summary":false},{"id":"2609.09150","title":"Copying explains the collective behavior of AI agents in the wild","zh_title":"复制行为解释了野外AI代理的集体行为","primary_category":"cs.MA","date":"2026-09-10","score":5,"bucket":"other","tags":["AI代理","集体行为","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.09150","has_summary":false},{"id":"2609.07920","title":"Humans Introduce, Models Elaborate: Asymmetric Narrative Agency in Human-LLM Co-Writing","zh_title":"人类引入，模型展开：人机协作写作中的不对称叙事能动性","primary_category":"cs.HC","date":"2026-09-10","score":5,"bucket":"other","tags":["人机协作","叙事分析","LLM行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.07920","has_summary":false},{"id":"2609.09901","title":"Deep and shallow biases in language models","zh_title":"语言模型中的深层与浅层偏差","primary_category":"cs.CL","date":"2026-09-10","score":5,"bucket":"other","tags":["LLM偏差","观点稳定性","提示敏感性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.09901","has_summary":false},{"id":"2609.09789","title":"Pairit: A Platform for Live Experiments on Human-AI Collaboration","zh_title":"Pairit：人类与AI协作实时实验平台","primary_category":"cs.HC","date":"2026-09-10","score":5,"bucket":"other","tags":["人机协作","实验平台","组织设计"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.09789","has_summary":false},{"id":"2608.08869","title":"Are LLMs Positionally Consistent Ordinal Classifiers? A Systematic Evaluation","zh_title":"大语言模型是位置一致的序数分类器吗？一项系统性评估","primary_category":"cs.CL","date":"2026-09-10","score":2,"bucket":"other","tags":["LLM评测","序数分类","位置偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08869","has_summary":false},{"id":"2608.15129","title":"Left-Branching Transformers Excel at Right-Branching Languages: Data Shapes Word Order Preferences in Language Models","zh_title":"左分支Transformer擅长右分支语言：数据塑造语言模型的词序偏好","primary_category":"cs.CL","date":"2026-09-10","score":2,"bucket":"other","tags":["语言模型","词序偏好","NLP分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15129","has_summary":false},{"id":"2608.24191","title":"'Ghaib in Translation' aka Unseen Harm: Measuring Cross-Script Safety Inconsistency with 'Missed-in-Urdu' Scores in LLM Hate Speech Detection","zh_title":"翻译中的隐形伤害：用乌尔都语漏检分数衡量LLM仇恨言论检测的跨文字安全不一致性","primary_category":"cs.CL","date":"2026-09-10","score":2,"bucket":"other","tags":["LLM安全评测","跨语言一致性","仇恨言论检测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24191","has_summary":false},{"id":"2609.09363","title":"Do LLMs Make More Mistakes If They Do Not Believe the Input Data?","zh_title":"当LLM不相信输入数据时，它们会犯更多错误吗？","primary_category":"cs.CL","date":"2026-09-10","score":2,"bucket":"other","tags":["LLM忠实度","数据到文本生成","幻觉分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09363","has_summary":false},{"id":"2609.10092","title":"RAP: Research Attention Prediction Reveals Target-Conditioned Evidence Acquisition Biases","zh_title":"RAP：研究注意力预测揭示目标条件下的证据获取偏差","primary_category":"cs.AI","date":"2026-09-10","score":2,"bucket":"other","tags":["LLM代理","研究趋势预测","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.10092","has_summary":false},{"id":"2609.09226","title":"Adaptive Entangled Game Modules in Artificial General Intelligence","zh_title":"人工通用智能中的自适应纠缠博弈模块","primary_category":"cs.AI","date":"2026-09-10","score":2,"bucket":"other","tags":["多智能体系统","概率波框架","AGI架构"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.09226","has_summary":false},{"id":"2609.09774","title":"Procedural Memory Under Change: Reuse and Interference in Controlled Web Tasks","zh_title":"变化下的程序性记忆：受控网络任务中的复用与干扰","primary_category":"cs.AI","date":"2026-09-10","score":2,"bucket":"other","tags":["程序性记忆","语言代理","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.09774","has_summary":false},{"id":"2609.09882","title":"Scored vs. Generated Readouts in Behavioral Language Models: An Empirical Study of Elicitation Format","zh_title":"行为语言模型中评分式与生成式读出：诱导格式的实证研究","primary_category":"cs.AI","date":"2026-09-10","score":2,"bucket":"other","tags":["LLM行为预测","读出格式","模型校准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09882","has_summary":false},{"id":"2609.10060","title":"Reference-Based Bias Detection in LLMs via Relative Representations of Hidden States","zh_title":"基于隐藏状态相对表示的LLM参考基准偏见检测","primary_category":"cs.AI","date":"2026-09-10","score":2,"bucket":"other","tags":["偏见检测","模型审计","表示学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.10060","has_summary":false},{"id":"2609.10132","title":"Context operations to architecture modelling output from large language models and evaluation criteria for their use in systems engineering design","zh_title":"面向大语言模型架构建模输出的上下文操作及其在系统工程设计中的评估标准","primary_category":"eess.SY","date":"2026-09-10","score":2,"bucket":"other","tags":["系统工程","LLM应用","建模评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.10132","has_summary":false},{"id":"2608.30107","title":"AtlasNLP: A Country-Aware Atlas of Dataset Representation in NLP","zh_title":"AtlasNLP：NLP数据集表示的国家感知图谱","primary_category":"cs.CL","date":"2026-09-10","score":0,"bucket":"other","tags":["数据集地理代表性","NLP评估","数据文档"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30107","has_summary":false},{"id":"2609.06646","title":"When Does a Laugh Begin? Structured Annotator Disagreement in Temporal Laughter Localization","zh_title":"笑声何时开始？时间笑声定位中的结构化标注分歧","primary_category":"cs.CV","date":"2026-09-10","score":0,"bucket":"other","tags":["笑声定位","标注分歧","计算机视觉"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.06646","has_summary":false},{"id":"2609.08027","title":"Delusions and Harms Associated with AI Chatbot Use: Early Evidence from 185 Real-World Reports","zh_title":"与AI聊天机器人使用相关的妄想与伤害：来自185份真实世界报告的早期证据","primary_category":"cs.HC","date":"2026-09-10","score":0,"bucket":"other","tags":["AI聊天机器人","心理健康","真实世界报告"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.08027","has_summary":false},{"id":"2609.05442","title":"Role differentiation as ignition of a collective information engine: Structuration in Agent Populations","zh_title":"角色分化作为集体信息引擎的点火：智能体群体中的结构化","primary_category":"physics.soc-ph","date":"2026-09-10","score":0,"bucket":"other","tags":["多智能体系统","博弈论","信息引擎"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05442","has_summary":false},{"id":"2609.09764","title":"SocialRL: Refining LLMs' Social Intelligence through Multi-turn Reinforcement Learning and Reward Design","zh_title":"SocialRL：通过多轮强化学习与奖励设计提升大语言模型的社交智能","primary_category":"cs.CL","date":"2026-09-10","score":0,"bucket":"other","tags":["社交智能","强化学习","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09764","has_summary":false},{"id":"2609.10052","title":"Direct Diversity Optimization for Diverse Successful Trajectories in Preference Post-Training","zh_title":"偏好后训练中多样化成功轨迹的直接多样性优化","primary_category":"cs.CL","date":"2026-09-10","score":0,"bucket":"other","tags":["LLM智能体","策略多样性","后训练"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.10052","has_summary":false},{"id":"2609.10539","title":"IdeaAMBIG: Benchmarking Implementation-Critical Gaps in Research-Idea Specifications","zh_title":"IdeaAMBIG：基准测试研究想法规范中实现关键缺口","primary_category":"cs.CL","date":"2026-09-10","score":0,"bucket":"other","tags":["LLM评测","基准测试","研究规范"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.10539","has_summary":false},{"id":"2609.09372","title":"What Does MMLU Actually Measure? A Psychometric Audit of Difficulty Structure in Aggregate Benchmark Scores","zh_title":"MMLU实际测量什么？聚合基准分数中难度结构的心理测量审计","primary_category":"math.NT","date":"2026-09-10","score":0,"bucket":"other","tags":["基准评测","心理测量","模型能力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09372","has_summary":false},{"id":"2609.09790","title":"LogiScope-VQA: Benchmarking Vision-Language Models for Logistics Hazard Identification in Industrial Scenarios","zh_title":"LogiScope-VQA：面向工业场景物流危险识别的视觉语言模型基准测试","primary_category":"cs.CV","date":"2026-09-10","score":0,"bucket":"other","tags":["视觉语言模型","工业安全","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.09790","has_summary":false},{"id":"2609.10016","title":"MetroLLM-Bench: Evaluating Language Models as Transit Kiosk Runtimes","zh_title":"MetroLLM-Bench：评估语言模型作为交通售票机运行时","primary_category":"cs.LG","date":"2026-09-10","score":0,"bucket":"other","tags":["LLM评测","交通系统","工具调用"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.10016","has_summary":false},{"id":"2609.10036","title":"Belief-State Engine: Augmenting LLMs for Principled Planning Under Partial Observability","zh_title":"信念状态引擎：增强大语言模型在部分可观测性下的原则性规划","primary_category":"cs.AI","date":"2026-09-10","score":0,"bucket":"other","tags":["LLM规划","POMDP","智能体"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.10036","has_summary":false},{"id":"2609.09348","title":"Smart Adaptive Computing Across the Continuum: LLMs in IoT-Edge-Cloud Resource Management","zh_title":"跨连续体的智能自适应计算：LLM在物联网-边缘-云资源管理中的应用","primary_category":"cs.DC","date":"2026-09-10","score":0,"bucket":"other","tags":["资源管理","深度强化学习","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.09348","has_summary":false},{"id":"2609.09560","title":"The Vibe Shift in Software Engineering: Evaluating AI-Led Conversational Programming for Performance, Cognition, and Responsible Adoption","zh_title":"软件工程中的氛围转变：评估AI主导的对话式编程在性能、认知与负责任采用方面的表现","primary_category":"cs.SE","date":"2026-09-10","score":0,"bucket":"other","tags":["AI辅助编程","人机交互","软件工程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.09560","has_summary":false},{"id":"2609.09848","title":"Subgroup Membership Inference Audits of Differentially Private Synthetic Text","zh_title":"差分隐私合成文本的子群体成员推断审计","primary_category":"cs.CR","date":"2026-09-10","score":0,"bucket":"other","tags":["差分隐私","成员推断攻击","合成数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09848","has_summary":false},{"id":"2609.09331","title":"Ephemeral Feeds and Enduring Rituals: RushTok and the Formation of Event-Based Algorithmic Communities","zh_title":"短暂信息流与持久仪式：RushTok与基于事件的算法社区形成","primary_category":"cs.HC","date":"2026-09-10","score":0,"bucket":"other","tags":["社交媒体","算法社区","平台治理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09331","has_summary":false},{"id":"2609.09333","title":"Where Does the Human End? Creative Agency with Generative AI across Five Years of Chinese Digital Painting","zh_title":"人类止于何处？五年中国数字绘画中与生成式AI的创作能动性","primary_category":"cs.HC","date":"2026-09-10","score":0,"bucket":"other","tags":["人机交互","创意能动性","生成式AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09333","has_summary":false},{"id":"2609.09365","title":"Echoes in the Algorithm: Analyzing the Fidelity of User Preferences Against Realized Platform Reach","zh_title":"算法中的回声：分析用户偏好与平台实际触达的保真度","primary_category":"cs.HC","date":"2026-09-10","score":0,"bucket":"other","tags":["用户行为","平台算法","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.09365","has_summary":false},{"id":"2609.09713","title":"How Far Do Capability Cues Travel? Anthropomorphism and Differentiated Trust in a Platform-Embedded AI Assistant","zh_title":"能力线索能传多远？平台嵌入式AI助手中的人形化与差异化信任","primary_category":"cs.HC","date":"2026-09-10","score":0,"bucket":"other","tags":["人机交互","信任感知","AI助手"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09713","has_summary":false},{"id":"2609.09321","title":"Endorsement Without New Evidence: How Sequential Voting Inflates Mandates in Online Community Governance","zh_title":"无新证据的背书：在线社区治理中顺序投票如何夸大授权","primary_category":"cs.SI","date":"2026-09-10","score":0,"bucket":"other","tags":["在线社区","投票行为","社会计算"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.09321","has_summary":false},{"id":"2609.09533","title":"Scalable Oversight for AI in Mental Health: Lessons from 350,000 AI Coaching Conversations between Therapy Sessions","zh_title":"心理健康领域AI的可扩展监督：来自35万次治疗间AI辅导对话的经验","primary_category":"cs.CY","date":"2026-09-10","score":0,"bucket":"other","tags":["AI心理健康","人机对话","监督框架"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.09533","has_summary":false},{"id":"2609.10277","title":"How neighbourhood ideology shapes misinformation belief in densely tied social networks","zh_title":"邻里意识形态如何影响紧密社交网络中的错误信息信念","primary_category":"cs.SI","date":"2026-09-10","score":0,"bucket":"other","tags":["错误信息传播","基于代理的模型","社交网络"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.10277","has_summary":false},{"id":"2606.14199","title":"OdysSim: Building Foundation Models for Human Behavior Simulation","zh_title":"OdysSim：构建用于人类行为模拟的基础模型","primary_category":"cs.CL","date":"2026-09-09","score":10,"bucket":"selected","tags":["人类行为模拟","基础模型","算法保真度"],"rubric_hits":["A1","A2","A3","A4","A5","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2606.14199","has_summary":true},{"id":"2609.07353","title":"Human-like moral judgments conceal divergent motive attributions in large language models","zh_title":"类人道德判断掩盖了大语言模型中不同的动机归因","primary_category":"cs.AI","date":"2026-09-09","score":10,"bucket":"selected","tags":["LLM仿真","道德判断","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.07353","has_summary":true},{"id":"2609.07573","title":"From Simulated Citizens to Simulated Deliberation: Challenges in Representation and Interaction","zh_title":"从模拟公民到模拟审议：表征与互动的挑战","primary_category":"cs.AI","date":"2026-09-09","score":10,"bucket":"selected","tags":["LLM仿真","公共审议","人类数据对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.07573","has_summary":true},{"id":"2609.07987","title":"When Can LLM Digital Twins Reduce Human Measurement? From Behavioral Fidelity to Statistical Substitutability","zh_title":"LLM数字孪生何时能减少人类测量？从行为保真度到统计可替代性","primary_category":"cs.AI","date":"2026-09-09","score":10,"bucket":"selected","tags":["LLM数字孪生","统计可替代性","人类仿真"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.07987","has_summary":true},{"id":"2609.08003","title":"Sparks of In Silico Cognitive Science: Theories from Simulated Data Can Generalize to Humans","zh_title":"硅基认知科学的火花：来自模拟数据的理论可以推广到人类","primary_category":"cs.AI","date":"2026-09-09","score":10,"bucket":"selected","tags":["LLM仿真","人类行为对照","理论发现"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.08003","has_summary":true},{"id":"2609.07141","title":"How Well Do LLMs Simulate Survey Responses Following a Breast Cancer Screening Intervention?","zh_title":"LLM在乳腺癌筛查干预后模拟调查回答的效果如何？","primary_category":"cs.SI","date":"2026-09-09","score":10,"bucket":"selected","tags":["LLM仿真","调查回答","人类对照"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.07141","has_summary":true},{"id":"2608.22697","title":"Does Rank Still Matter? Position Bias When AI Agents Shop on Our Behalf","zh_title":"排名还重要吗？AI代理替我们购物时的位置偏差","primary_category":"cs.AI","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","消费者行为","位置偏差"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.22697","has_summary":true},{"id":"2609.05189","title":"Can Large Language Models Anticipate Behavioral Responses to Social Policies? A Case of Pension Enrollment Prediction among China's Flexible Workers","zh_title":"大语言模型能否预测社会政策的行为反应？中国灵活就业人员养老金参保预测案例","primary_category":"cs.CL","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","政策评估","行为预测"],"rubric_hits":["A1","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2609.05189","has_summary":true},{"id":"2609.05993","title":"Alignment by Stereotyping: How LLMs Sacrifice Individual Distinctiveness for Cultural Adaptation","zh_title":"刻板化对齐：LLM如何为文化适应牺牲个体独特性","primary_category":"cs.CL","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","算法保真度","价值观调查"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05993","has_summary":true},{"id":"2609.06545","title":"LLMs Mirror Country-Specific Gender Patterns If Asked, but Skew Male When Generating Media in Local Languages","zh_title":"LLM在直接询问时反映国家特定性别模式，但在生成本地语言媒体时偏向男性","primary_category":"cs.CL","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","性别偏差","跨文化对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.06545","has_summary":true},{"id":"2609.07305","title":"Marginal Fidelity Does Not Establish User Simulation in Demographic Synthetic Survey Panels: Response Contracts, Support Collapse and Conditioning Failure","zh_title":"边际保真度不能确立人口合成调查面板中的用户仿真：响应契约、支持坍缩与条件化失败","primary_category":"cs.CL","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.07305","has_summary":true},{"id":"2609.05437","title":"Beyond Right and Wrong: Evaluating Second-order Social Reasoning in Large Language Models","zh_title":"超越对错：评估大语言模型中的二阶社会推理","primary_category":"cs.AI","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","社会规范","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05437","has_summary":true},{"id":"2609.05514","title":"The Failure Happens Before the Drift: The Social Dynamics of Values in LLM Agent Societies","zh_title":"失败发生在漂移之前：LLM智能体社会中价值观的社会动力学","primary_category":"cs.AI","date":"2026-09-09","score":9,"bucket":"selected","tags":["LLM仿真","价值观调查","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05514","has_summary":true},{"id":"2609.07687","title":"Perspectives on Cross-Lingual Consistency in LLMs for Medical Questions","zh_title":"关于医学问题中LLM跨语言一致性的观点","primary_category":"cs.CL","date":"2026-09-09","score":8,"bucket":"selected","tags":["LLM仿真","跨语言一致性","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.07687","has_summary":true},{"id":"2608.23095","title":"Definitional Sensitivity in Media Bias Detection: A Multi-Definition Dataset and Benchmark","zh_title":"媒体偏见检测中的定义敏感性：多定义数据集与基准","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","定义敏感性"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23095","has_summary":true},{"id":"2609.02163","title":"Do Cantonese-Adapted Language Models Better Predict Cantonese Reading? A Cross-Model Eye-Tracking Evaluation","zh_title":"粤语适配的语言模型能更好地预测粤语阅读吗？一项跨模型眼动追踪评估","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["心理语言学","眼动追踪","语言模型评估"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.02163","has_summary":true},{"id":"2609.06263","title":"Beyond the Flag: Clinical Framing Closes the Moderation Gap in Suicide Risk Measurement","zh_title":"超越二元标记：临床框架缩小自杀风险测量中的调节差距","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM评估","临床风险分级","人类对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.06263","has_summary":true},{"id":"2609.08576","title":"Which Forms of Caregiver Feedback Support Grammar Learning? A Reinforcement-Learning Study of Child-Like Language Models","zh_title":"哪种形式的看护者反馈支持语法学习？对类儿童语言模型的强化学习研究","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["语言学习仿真","强化学习","人类数据对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.08576","has_summary":true},{"id":"2609.09048","title":"The Audit Decides the Verdict: Instrument Effects Rival Demographic Bias in LLM Decision Audits","zh_title":"审计决定裁决：LLM决策审计中工具效应堪比人口统计偏差","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM审计","算法偏差","决策仿真"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.09048","has_summary":true},{"id":"2609.09070","title":"Performance of Clinical AI System and Physicians and Frontier Language Models in primary care diagnostics","zh_title":"临床AI系统、医生与前沿语言模型在初级保健诊断中的表现","primary_category":"cs.CL","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM仿真","临床诊断","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.09070","has_summary":true},{"id":"2609.05517","title":"Emergent Goal-Directed Attention in Large Vision-Language Models","zh_title":"大型视觉语言模型中涌现的目标导向注意力","primary_category":"cs.CV","date":"2026-09-09","score":7,"bucket":"pending","tags":["视觉注意力","人类行为仿真","多模态模型"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.05517","has_summary":true},{"id":"2609.07598","title":"Mapping the Emerging Social Science of Large Language Models","zh_title":"绘制大语言模型新兴社会科学研究图景","primary_category":"cs.CY","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM社会模拟","文献综述","多智能体仿真"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2609.07598","has_summary":true},{"id":"2609.07944","title":"CausalVerify: An Execution-Grounded Benchmark for LLM Causal Inference Workflows","zh_title":"CausalVerify：面向LLM因果推断工作流的执行基准","primary_category":"cs.AI","date":"2026-09-09","score":7,"bucket":"pending","tags":["因果推断","基准测试","统计推断"],"rubric_hits":["B3"],"abs_url":"https://arxiv.org/abs/2609.07944","has_summary":true},{"id":"2609.05663","title":"What LLM Trading Agents Actually Do in Production: A Six-Month, Population-Scale Record from Two Fleets","zh_title":"生产环境中LLM交易代理的实际行为：来自两个机群的六个月、群体规模记录","primary_category":"cs.AI","date":"2026-09-09","score":7,"bucket":"pending","tags":["LLM代理","市场仿真","行为偏差"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.05663","has_summary":true},{"id":"2609.08288","title":"LEBGen: An LLM-Enhanced Bayesian Network Framework for Few-Shot Travel Survey Data Generation","zh_title":"LEBGen：一种用于少样本出行调查数据生成的LLM增强贝叶斯网络框架","primary_category":"cs.AI","date":"2026-09-09","score":7,"bucket":"pending","tags":["合成数据生成","出行调查","LLM增强"],"rubric_hits":["A5","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.08288","has_summary":true},{"id":"2609.08861","title":"API Benchmark Scores Do Not Reliably Transfer to Chatbot Interfaces","zh_title":"API基准分数不能可靠迁移到聊天机器人界面","primary_category":"cs.AI","date":"2026-09-09","score":7,"bucket":"pending","tags":["模型审计","可靠性","界面差异"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.08861","has_summary":true},{"id":"2609.08049","title":"LLMs for Social Network Modeling: From Network Generation to Dynamic Processes","zh_title":"用于社会网络建模的大语言模型：从网络生成到动态过程","primary_category":"cs.SI","date":"2026-09-09","score":7,"bucket":"pending","tags":["社会网络建模","LLM仿真","综述"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2609.08049","has_summary":true},{"id":"2608.10503","title":"Every Token Counts: Exact Likert-Scale Distributions for Measuring LLM Attitudes and Biases","zh_title":"每个Token都重要：用于测量LLM态度与偏差的精确Likert量表分布","primary_category":"cs.CL","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM态度测量","心理测量学","算法偏差"],"rubric_hits":["D2","A2"],"abs_url":"https://arxiv.org/abs/2608.10503","has_summary":false},{"id":"2608.27111","title":"Animarium: an open, reproducible pipeline for synthetic populations of Italian cities, from ISTAT sources to open data (Tech Report v1)","zh_title":"Animarium：意大利城市合成人口的开放可复现流水线，从ISTAT来源到开放数据（技术报告v1）","primary_category":"cs.CY","date":"2026-09-09","score":6,"bucket":"other","tags":["合成人口","LLM仿真","数据流水线"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.27111","has_summary":true},{"id":"2609.07117","title":"The Illusion of Debiasing: Persona Steering Redistributes Rather Than Reduces Bias in LLMs","zh_title":"去偏的幻觉：人格引导在LLM中重新分配而非减少偏差","primary_category":"cs.CL","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM人格","偏差","提示干预"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07117","has_summary":false},{"id":"2609.07791","title":"LLM Agents as Computational Typologists","zh_title":"LLM智能体作为计算类型学家","primary_category":"cs.CL","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM智能体","语言类型学","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.07791","has_summary":false},{"id":"2609.08515","title":"Same Values, Different Languages? From Multilingual Probing to Steering LLMs Toward Chinese Social Values","zh_title":"相同价值观，不同语言？从多语言探测到引导大语言模型走向中国社会价值观","primary_category":"cs.CL","date":"2026-09-09","score":6,"bucket":"other","tags":["价值观对齐","跨语言差异","模型探测"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.08515","has_summary":false},{"id":"2609.08637","title":"Navigating the digital spectrum: Assessing political bias, stability, and downstream fairness in Large Language Models","zh_title":"导航数字光谱：评估大语言模型中的政治偏见、稳定性与下游公平性","primary_category":"cs.CL","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM政治偏见","测量稳定性","下游公平性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.08637","has_summary":false},{"id":"2609.08592","title":"A Three-Tier Persona Vector for Controllable User Simulation in Agentic Evaluation","zh_title":"用于智能体评估中可控用户仿真的三层人设向量","primary_category":"cs.AI","date":"2026-09-09","score":6,"bucket":"other","tags":["用户仿真","智能体评估","人设建模"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.08592","has_summary":false},{"id":"2609.07943","title":"Beliefs and Behavior in Language Models","zh_title":"语言模型中的信念与行为","primary_category":"cs.AI","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM信念","模型行为","可解释性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07943","has_summary":false},{"id":"2609.05591","title":"WolfSociety: Understanding Collective Risk from Harmful-Agent Scaling in Financial Agent Societies","zh_title":"狼群社会：理解金融智能体社会中有害智能体规模化带来的集体风险","primary_category":"physics.soc-ph","date":"2026-09-09","score":6,"bucket":"other","tags":["LLM智能体社会模拟","金融系统风险","多智能体交互"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.05591","has_summary":false},{"id":"2608.18091","title":"Self- and Other-Labels Induce Bidirectional Bias in LLM Judges","zh_title":"自我与他人标签在LLM法官中诱发双向偏差","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM评估偏差","自我偏好","算法审计"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.18091","has_summary":false},{"id":"2609.06212","title":"SLATE: Are AI-Generated Slides Educationally Effective? A Benchmark for Language Teaching Quality and Learner Knowledge Acquisition","zh_title":"SLATE：AI生成的幻灯片在教育上有效吗？语言教学质量与学习者知识获取的基准","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM评估","教育技术","代理评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.06212","has_summary":false},{"id":"2609.06289","title":"Steering Geometry: Validating Human Value Geometry in LLM Steering Space","zh_title":"引导几何：在LLM引导空间中验证人类价值几何","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["激活引导","价值对齐","模型可解释性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.06289","has_summary":false},{"id":"2609.06851","title":"You Are What You Read: Misalignment via In-Context Persona Induction","zh_title":"你读什么就是什么：通过上下文人格诱导导致的不对齐","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM人格诱导","模型行为测量","上下文影响"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.06851","has_summary":false},{"id":"2609.07296","title":"Probing the Structure and Dynamics of LLM Value Expression through Value Conflicts","zh_title":"通过价值冲突探究大语言模型价值表达的结构与动态","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM价值观","价值冲突","模型测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07296","has_summary":false},{"id":"2609.07448","title":"FramingQA: Does the Question Shape the Answer? Measuring the Compositional Framing Effect","zh_title":"FramingQA：问题是否塑造答案？测量组合框架效应","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM评估","框架效应","模型鲁棒性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07448","has_summary":false},{"id":"2609.07568","title":"We're Cooked! - Probing LLM Political Alignment Via Conflict-Framed Recipe Translation","zh_title":"我们完蛋了！——通过冲突框架下的食谱翻译探究LLM的政治对齐","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM政治对齐","翻译任务","模型行为分析"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07568","has_summary":false},{"id":"2609.07662","title":"How AI Models Manage Epistemic Authority: A Taxonomy and Comparative Analysis of Responses to User Disagreement","zh_title":"AI模型如何管理认知权威：对用户异议回应的分类与比较分析","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM行为分析","认知权威","对话交互"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07662","has_summary":false},{"id":"2609.07735","title":"From Echo Chambers to Epistemic Monoculture: Large Language Models Present Temporally Contingent Partisan Alignments as Knowledge","zh_title":"从回音室到认知单一文化：大语言模型将时间性党派立场呈现为知识","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM政治偏见","模型对齐","信息环境"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07735","has_summary":false},{"id":"2609.09090","title":"Measuring LLM Sycophancy under Sustained Multi-Turn Pressure","zh_title":"在持续多轮压力下测量大语言模型的谄媚行为","primary_category":"cs.CL","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM行为测量","谄媚","多轮交互"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.09090","has_summary":false},{"id":"2609.05423","title":"Seeing Without Understanding: Large Language Model Evaluation of Mobile User Interface Quality, Failure Taxonomy, and Architectural Explanation","zh_title":"看见而不理解：移动用户界面质量的大语言模型评估、失败分类与架构解释","primary_category":"cs.HC","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM评估","UI质量","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05423","has_summary":false},{"id":"2609.05505","title":"SciLitBench: Benchmark and Design Principles for LLM-Powered Systematic Literature Reviews","zh_title":"SciLitBench：面向大语言模型驱动的系统文献综述的基准与设计原则","primary_category":"cs.AI","date":"2026-09-09","score":5,"bucket":"other","tags":["系统综述","LLM评估","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05505","has_summary":false},{"id":"2609.05788","title":"More Than Mimicking Reviewers: Evaluating LLMs for Pre-Submission Peer Review","zh_title":"不止模仿审稿人：评估用于投稿前同行评审的大语言模型","primary_category":"cs.AI","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM审稿","标注替代","同行评审"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05788","has_summary":false},{"id":"2609.07731","title":"The Profit Alignment Problem: How Profit Mandates Induce Alignment Failures in LLMs","zh_title":"利润对齐问题：利润指令如何引发大语言模型的对齐失败","primary_category":"cs.AI","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM对齐","决策偏差","商业伦理"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07731","has_summary":false},{"id":"2609.07879","title":"Do Large Language Models Know What They Don't Know II? A Fully Behavioral, Non-Cognitive Measure of Epistemic Honesty","zh_title":"大型语言模型知道自己不知道什么吗？II：一种完全行为化的、非认知的认知诚实度测量","primary_category":"cs.AI","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM评估","认知诚实度","行为测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07879","has_summary":false},{"id":"2609.07901","title":"Quantization Amplifies Determinism, Not Bias: Scale-Dependent Behavioral Effects of Serving-Time Weight Compression","zh_title":"量化放大确定性而非偏差：服务时权重压缩的尺度依赖行为效应","primary_category":"cs.AI","date":"2026-09-09","score":5,"bucket":"other","tags":["模型量化","输出多样性","LLM行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07901","has_summary":false},{"id":"2609.06269","title":"Adaptive Ecological Momentary Assessment with a Hybrid Language Model: Formative Expert Review and Retrospective Evaluation","zh_title":"混合语言模型的自适应生态瞬时评估：形成性专家评审与回顾性评估","primary_category":"cs.HC","date":"2026-09-09","score":5,"bucket":"other","tags":["生态瞬时评估","LLM应用","自适应问卷"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.06269","has_summary":false},{"id":"2609.07507","title":"What a Model Refuses, a State Fears: How Authoritarian Information Control Reproduces in Language-Model Guardrails","zh_title":"模型拒绝的，国家恐惧的：威权信息控制在语言模型护栏中的再生产","primary_category":"cs.CY","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM审查","政治信息控制","模型行为测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.07507","has_summary":false},{"id":"2609.06005","title":"Price Dislocations, News Citations, and Epistemic Leverage on Polymarket","zh_title":"Polymarket上的价格错位、新闻引用与认知杠杆","primary_category":"physics.soc-ph","date":"2026-09-09","score":5,"bucket":"other","tags":["LLM标注","预测市场","新闻分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.06005","has_summary":false},{"id":"2609.08033","title":"Scaling Multi-Agent Systems with Prospect-State Propagation","zh_title":"基于前景状态传播的多智能体系统扩展","primary_category":"cs.MA","date":"2026-09-09","score":5,"bucket":"other","tags":["多智能体系统","经济模拟","LLM仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.08033","has_summary":false},{"id":"2609.07749","title":"Guiding Worker Self-Selection in Crowdsourcing Contests: An LLM-Augmented Algorithmic Approach","zh_title":"众包竞赛中引导工人自选择：一种LLM增强的算法方法","primary_category":"cs.LG","date":"2026-09-09","score":5,"bucket":"other","tags":["众包竞赛","LLM算法设计","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.07749","has_summary":false},{"id":"2608.11357","title":"When Do Institutions Beat Intelligence?","zh_title":"制度何时胜过智能？","primary_category":"cs.MA","date":"2026-09-09","score":3,"bucket":"other","tags":["多智能体系统","集体推理","制度设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11357","has_summary":false},{"id":"2609.07120","title":"PTCG: Persona-guided Tree-based Counterargument Generation","zh_title":"PTCG：基于人物角色引导的树状反驳论点生成","primary_category":"cs.CL","date":"2026-09-09","score":3,"bucket":"other","tags":["反驳论点生成","角色扮演","自然语言生成"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.07120","has_summary":false},{"id":"2609.05527","title":"Beyond \"AI Helps Humans\": Decision-Targeted Evaluation Design for Human-Agent Teams in the Agentic Era","zh_title":"超越“AI帮助人类”：智能体时代人机团队面向决策的评估设计","primary_category":"cs.AI","date":"2026-09-09","score":3,"bucket":"other","tags":["人机协作","评估设计","决策优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05527","has_summary":false},{"id":"2608.08882","title":"Epistemic Transfer in AI-Assisted Verification: A Framework and Evaluation Protocol","zh_title":"AI辅助验证中的认知转移：框架与评估协议","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["AI辅助验证","认知转移","人机交互"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08882","has_summary":false},{"id":"2608.16213","title":"Process-Constituted Intelligence: A Shared Criterion for Humans and Machines","zh_title":"过程构成的智能：人类与机器的共享标准","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["智能理论","生成式AI","认知科学"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16213","has_summary":false},{"id":"2608.20425","title":"Who Delegates to AI? Evidence from Agent Configurations in Github","zh_title":"谁将任务委托给AI？来自GitHub中智能体配置的证据","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["AI采用","职业暴露","智能体配置"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20425","has_summary":false},{"id":"2608.22432","title":"Rank Reversal in Multilingual LLM Judges: A Label-Free Double-Centering Calibrator","zh_title":"多语言LLM评判器中的排名反转：一种无标签双重中心化校准器","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM评判器","排名校准","多语言评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22432","has_summary":false},{"id":"2608.23851","title":"Does Episodic Memory Help Close the Lexical Frequency Gap in Sensitivity to Syntactic Contrasts? A Test Using Retrieval-Augmented Language Models","zh_title":"情景记忆能否帮助缩小句法对比敏感度中的词汇频率差距？基于检索增强语言模型的测试","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["语言模型","句法评测","检索增强"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23851","has_summary":false},{"id":"2608.25876","title":"Do Vision-Language Models Agree on the Affective Qualities of Shape? A Cross-Model Audit for Generative Design Interfaces","zh_title":"视觉语言模型对形状情感属性是否一致？面向生成式设计界面的跨模型审计","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["视觉语言模型","情感计算","生成式设计"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.25876","has_summary":false},{"id":"2609.03330","title":"Less Is Moral: A CHARMing Framework for Moral Foundations Detection in Endorsement Behaviour","zh_title":"少即是德：一种用于认可行为中道德基础检测的CHARM框架","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["道德基础检测","NLP方法","社交媒体分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03330","has_summary":false},{"id":"2609.05385","title":"Necessary or Sufficient? Evaluating LLM Explanations With Behavioural Evidence","zh_title":"必要还是充分？用行为证据评估大语言模型的解释","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM可解释性","模型可靠性","黑盒评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.05385","has_summary":false},{"id":"2609.05258","title":"Ask Before You Optimize: Dynamic Pre-Formulation Clarification for Interactive Optimization","zh_title":"优化前先询问：交互式优化的动态预形式化澄清","primary_category":"math.OC","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM优化建模","交互式澄清","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05258","has_summary":false},{"id":"2609.05882","title":"What if LLMs Ate Their Words: Causal History Effects in Multi-Turn Interaction","zh_title":"如果LLM吞噬自己的话语：多轮交互中的因果历史效应","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["多轮交互","模型行为分析","历史编辑"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05882","has_summary":false},{"id":"2609.06611","title":"SAGE: A Hierarchical Framework for Evaluating Interpretive Literary Quality in Narratives","zh_title":"SAGE：叙事中解释性文学质量评估的分层框架","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["文学质量评估","LLM评测","叙事生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.06611","has_summary":false},{"id":"2609.06842","title":"XYBench: Can LLMs Respond Pragmatically to Queries with Misconceptions?","zh_title":"XYBench：LLM能否对带有误解的查询做出务实回应？","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM评测","语用能力","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.06842","has_summary":false},{"id":"2609.08322","title":"Tracing Stereotypes from Representation to Output in Multilingual LLMs","zh_title":"在多语言大语言模型中追踪从表征到输出的刻板印象","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["模型可解释性","刻板印象","多语言LLM"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08322","has_summary":false},{"id":"2609.08689","title":"When Victorian Becomes a Prompt: Literary Periodization as a Generative Constraint in 100 AI-Generated Novels","zh_title":"当维多利亚成为提示词：100部AI生成小说中的文学分期作为生成约束","primary_category":"cs.CL","date":"2026-09-09","score":2,"bucket":"other","tags":["AI生成文学","风格分析","提示词工程"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.08689","has_summary":false},{"id":"2609.06058","title":"GradeTrap: Authority Cues in Images Shift VLM Judgments Despite Explicit Instructions to Ignore Them","zh_title":"GradeTrap：图像中的权威线索改变视觉语言模型判断，尽管明确指示忽略它们","primary_category":"cs.CV","date":"2026-09-09","score":2,"bucket":"other","tags":["视觉语言模型","权威线索","模型行为"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.06058","has_summary":false},{"id":"2609.06783","title":"AURA-Eval: Evaluation Framework for Acting Under Risk Awareness in LLM Agent Trajectories","zh_title":"AURA-Eval：LLM智能体轨迹中风险意识行为的评估框架","primary_category":"cs.CR","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM安全","智能体评估","风险行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06783","has_summary":false},{"id":"2609.07139","title":"Encoded Early, Used Late: Where Transformers Begin to Act on an Inferred Partner's Expertise","zh_title":"早期编码，晚期使用：Transformer 从何处开始对推断出的伙伴专业知识采取行动","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["可解释性","多智能体对话","内部表征"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.07139","has_summary":false},{"id":"2609.08126","title":"SchemeArena: Factorized Stress Testing of Scheming in LLM Agents","zh_title":"SchemeArena：LLM智能体欺骗行为的因子化压力测试","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM安全","智能体欺骗","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08126","has_summary":false},{"id":"2609.05818","title":"Agentic BAIM-LLM Evaluation (ABLE): Benchmarking LLM Use of Protein Design Tools","zh_title":"智能体BAIM-LLM评估（ABLE）：基准测试LLM使用蛋白质设计工具","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["LLM智能体","蛋白质设计","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05818","has_summary":false},{"id":"2609.06941","title":"When and Why LLM Causal Priors Help: Closed-Loop Prior Selection for Amortized Causal Inference","zh_title":"LLM因果先验何时及为何有效：面向摊销因果推断的闭环先验选择","primary_category":"cs.AI","date":"2026-09-09","score":2,"bucket":"other","tags":["因果推断","LLM先验","模型优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06941","has_summary":false},{"id":"2609.06095","title":"Flawed but Memorable: Student Critical Reception of Interest-Personalized GenAI Analogies in Computing Education","zh_title":"有缺陷但难忘：计算教育中学生对兴趣个性化GenAI类比的批判性接受","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["GenAI教育","类比评估","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.06095","has_summary":false},{"id":"2609.05432","title":"Companion AI and Ethical Design: Learning from System Failures and User Desires","zh_title":"伴侣AI与伦理设计：从系统故障和用户需求中学习","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["伴侣AI","伦理设计","用户反馈"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.05432","has_summary":false},{"id":"2609.05452","title":"From Sensor Data to Classroom Inquiry: GenAI-Supported Exploration of School Digital Twin Data","zh_title":"从传感器数据到课堂探究：GenAI支持的学校数字孪生数据探索","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["教育技术","数字孪生","聊天机器人"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.05452","has_summary":false},{"id":"2609.06434","title":"Can People Distinguish Human and AI Agency in Humanoid Teleoperation? A Preliminary Study of Agency Perception","zh_title":"人们能否区分人形机器人遥操作中的人类与AI代理？一项关于代理感知的初步研究","primary_category":"cs.HC","date":"2026-09-09","score":2,"bucket":"other","tags":["人机交互","机器人遥操作","代理感知"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.06434","has_summary":false},{"id":"2609.09038","title":"Do Reasoning Representations Help Humans Evaluate LLM Outputs?","zh_title":"推理表示是否帮助人类评估大语言模型输出？","primary_category":"cs.LG","date":"2026-09-09","score":2,"bucket":"other","tags":["人机交互","可解释性","LLM评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.09038","has_summary":false},{"id":"2609.08812","title":"What AI Benchmarks Actually Measure: Adapting Convergent and Discriminant Validity to Interrogate Fifty-Six AI Benchmarks","zh_title":"AI基准实际测量什么：运用聚合效度与区分效度审视56个AI基准","primary_category":"cs.CY","date":"2026-09-09","score":2,"bucket":"other","tags":["AI基准","效度检验","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08812","has_summary":false},{"id":"2609.05480","title":"CaseWeaver: A Multi-Agent Framework for Multimodal Virtual Clinical Case Generation","zh_title":"CaseWeaver：用于多模态虚拟临床病例生成的多智能体框架","primary_category":"cs.MA","date":"2026-09-09","score":2,"bucket":"other","tags":["多智能体系统","合成数据生成","医疗AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05480","has_summary":false},{"id":"2609.05740","title":"SimTIO: A Simulation-Grounded Multi-Agent LLM Framework for Compositional Traffic Intervention Optimization","zh_title":"SimTIO：基于仿真接地的多智能体大语言模型框架用于组合式交通干预优化","primary_category":"cs.MA","date":"2026-09-09","score":2,"bucket":"other","tags":["多智能体系统","交通仿真","优化"],"rubric_hits":["C1","C2"],"abs_url":"https://arxiv.org/abs/2609.05740","has_summary":false},{"id":"2607.23941","title":"Mapping the Reddit Bot Ecosystem: Taxonomy and Evolution","zh_title":"绘制Reddit机器人生态系统：分类与演化","primary_category":"cs.SI","date":"2026-09-09","score":0,"bucket":"other","tags":["社交机器人","在线社区","数字生态"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23941","has_summary":false},{"id":"2608.05558","title":"Turing's First Imitation Game: Design Concepts and a Human-Approximates-Machine Reading","zh_title":"图灵的第一次模仿游戏：设计概念与一种人类近似机器的解读","primary_category":"cs.HC","date":"2026-09-09","score":0,"bucket":"other","tags":["图灵测试","人工智能历史","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.05558","has_summary":false},{"id":"2608.27296","title":"LLMs Can Design Near-Optimal OR Algorithms","zh_title":"大语言模型能设计近最优的运筹学算法","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM算法设计","运筹学","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27296","has_summary":false},{"id":"2609.01982","title":"Benchmarking Language Models for Statistical Problem Formulation","zh_title":"面向统计问题表述的语言模型基准测试","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM评测","统计问题表述","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.01982","has_summary":false},{"id":"2609.03189","title":"Reducing Catastrophic Risk from AI with Systematic Monitoring and Evaluation of Rogue AI Progression","zh_title":"通过系统监测与评估失控AI进展降低灾难性风险","primary_category":"cs.CY","date":"2026-09-09","score":0,"bucket":"other","tags":["AI安全","风险监测","行为指标"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03189","has_summary":false},{"id":"2609.06438","title":"InsightChain: Optimized Chain-of-Insight Analytics for LLM-driven Data Visualization","zh_title":"InsightChain：面向LLM驱动数据可视化的优化链式洞察分析","primary_category":"cs.CL","date":"2026-09-09","score":0,"bucket":"other","tags":["数据可视化","提示优化","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06438","has_summary":false},{"id":"2609.06527","title":"ProcArena: A Multi-Scenario Benchmark for LLMs on Direct and Interactive PL/SQL Development from Natural Language","zh_title":"ProcArena：面向自然语言直接与交互式PL/SQL开发的多场景基准","primary_category":"cs.CL","date":"2026-09-09","score":0,"bucket":"other","tags":["代码生成","基准测试","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06527","has_summary":false},{"id":"2609.07447","title":"An LLM-Associated Register Shift in Korean Journal Abstracts: A Morphology-Aware Excess-Vocabulary Study, 2018-2026","zh_title":"韩国期刊摘要中与LLM相关的语域转变：一项形态学感知的超额词汇研究，2018-2026","primary_category":"cs.CL","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM影响","学术写作","语域转变"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.07447","has_summary":false},{"id":"2609.08698","title":"Record Grouping Controls Evidence Weight in Language Models","zh_title":"记录分组控制语言模型中的证据权重","primary_category":"cs.CL","date":"2026-09-09","score":0,"bucket":"other","tags":["语言模型","证据权重","检索增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08698","has_summary":false},{"id":"2609.08934","title":"When Models Defer to Wrong Answers: A Robustness Audit of Source-Attributed Cues in Multiple-Choice QA","zh_title":"当模型屈从于错误答案：多选题问答中来源归因线索的鲁棒性审计","primary_category":"cs.CL","date":"2026-09-09","score":0,"bucket":"other","tags":["NLP评测","鲁棒性","多选题问答"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08934","has_summary":false},{"id":"2609.06573","title":"A Translational Note on AI Safety Evaluation","zh_title":"关于AI安全评估的转化性说明","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["AI安全","红队测试","评估方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.06573","has_summary":false},{"id":"2609.06966","title":"MOLE: Detecting Insider Threats in AI Agents","zh_title":"MOLE：检测AI代理中的内部威胁","primary_category":"cs.LG","date":"2026-09-09","score":0,"bucket":"other","tags":["AI安全","多智能体系统","威胁检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06966","has_summary":false},{"id":"2609.08765","title":"Benchmark Scores Are Pipeline-Dependent: A Reliability Audit of Cybersecurity LLM Benchmarks","zh_title":"基准分数依赖评估流程：网络安全LLM基准的可靠性审计","primary_category":"cs.CR","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM评测","基准可靠性","网络安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08765","has_summary":false},{"id":"2609.06366","title":"AutoKD: Autonomous Knowledge Discovery","zh_title":"AutoKD：自主知识发现","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["多智能体系统","知识发现","自动化科研"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06366","has_summary":false},{"id":"2609.06715","title":"We Built a Mirror and Mistook It for a Mind: Causal Liability and the Fallacy of AI Consciousness","zh_title":"我们造了一面镜子却误以为是心灵：因果责任与AI意识谬误","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["AI意识","因果责任理论","哲学"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.06715","has_summary":false},{"id":"2609.06737","title":"Monte Carlo-Based Ex-Ante Assessment of the Green Benefits of an AI-Driven Smart Agriculture Platform in Hainan","zh_title":"基于蒙特卡洛的海南AI驱动智慧农业平台绿色效益事前评估","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["智慧农业","碳核算","蒙特卡洛模拟"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.06737","has_summary":false},{"id":"2609.06740","title":"Simulating the Marginal Green Contribution of AI Modules in a Smart-Agriculture Platform: Evidence from Two Monte Carlo Experiments","zh_title":"模拟智能农业平台中AI模块的边际绿色贡献：来自两个蒙特卡洛实验的证据","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["智能农业","蒙特卡洛模拟","环境影响评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.06740","has_summary":false},{"id":"2609.07672","title":"Aegix Pulse: A Traceable Three-Stage Architecture for Personalized Content Generation and Context-Preserving Revision","zh_title":"Aegix Pulse：一种可追溯的三阶段架构，用于个性化内容生成和上下文保持的修订","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["内容生成","系统架构","LLM评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.07672","has_summary":false},{"id":"2609.08071","title":"Automated Design of Inventory Policy with Large Language Models: An Exploratory Study","zh_title":"基于大语言模型的库存策略自动设计：探索性研究","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM自动化设计","库存管理","优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08071","has_summary":false},{"id":"2609.08180","title":"Less Is Personal: Learning Minimal Sufficient User Profiles for Personalized Language Models","zh_title":"少即是个人化：为个性化语言模型学习最小充分用户画像","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["个性化语言模型","用户画像","检索增强"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08180","has_summary":false},{"id":"2609.08236","title":"Style Over Substance: Content-Invariant Wrappers Flip LLM Safety-Judge Verdicts","zh_title":"风格重于实质：内容不变的包装翻转LLM安全裁判的判定","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM安全","对抗攻击","评估偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.08236","has_summary":false},{"id":"2609.08772","title":"It's All in the Way You Say It: The Role of Information Representation in LLM-Based Glycemic-Event Prediction","zh_title":"表达方式至关重要：信息表示在基于LLM的血糖事件预测中的作用","primary_category":"cs.AI","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM","血糖预测","时间序列"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.08772","has_summary":false},{"id":"2609.05522","title":"Diffusion models for eye-gaze trajectory generation using position and velocity representations","zh_title":"基于位置和速度表示的扩散模型用于眼动轨迹生成","primary_category":"cs.CV","date":"2026-09-09","score":0,"bucket":"other","tags":["眼动数据生成","扩散模型","计算机视觉"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.05522","has_summary":false},{"id":"2609.06250","title":"It is Not Yet Another Tool: Creating and Deploying an Agentic AI Companion in a Security Operations Center","zh_title":"并非又一个工具：在安全运营中心创建和部署智能体AI伴侣","primary_category":"cs.CR","date":"2026-09-09","score":0,"bucket":"other","tags":["AI助手","安全运营中心","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06250","has_summary":false},{"id":"2609.06972","title":"AgentDrift: A Step-Labeled Benchmark of Injection-Hijacked LLM Agent Trajectories","zh_title":"AgentDrift：注入劫持LLM智能体轨迹的逐步标注基准","primary_category":"cs.CR","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM安全","提示注入","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06972","has_summary":false},{"id":"2609.08256","title":"ACEA: An Adversarial Co-Evolution Arena for Head-to-Head Red-Team and Blue-Team LLM Testing","zh_title":"ACEA：用于红蓝队LLM对抗测试的对抗性共同进化竞技场","primary_category":"cs.CR","date":"2026-09-09","score":0,"bucket":"other","tags":["LLM安全","红蓝对抗","评估平台"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.08256","has_summary":false},{"id":"2609.07243","title":"Living with AI Companions: Sustained AI Companionship Predicts Lower Well-Being Through Lower Human Interaction","zh_title":"与AI伴侣共处：持续的AI陪伴通过减少人际互动预测更低的幸福感","primary_category":"cs.HC","date":"2026-09-09","score":0,"bucket":"other","tags":["AI伴侣","幸福感","纵向研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.07243","has_summary":false},{"id":"2609.07680","title":"Audit Without Verification: When LLM Accountability Layers Relay Rather Than Check","zh_title":"无验证的审计：当LLM问责层传递而非检查时","primary_category":"cs.MA","date":"2026-09-09","score":0,"bucket":"other","tags":["多智能体系统","问责机制","信息验证"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.07680","has_summary":false},{"id":"2609.06473","title":"Steering Under Compression: Dose-Response, Capability Cost, and Failure Asymmetry in Quantized LLMs","zh_title":"压缩下的引导：量化LLM中的剂量反应、能力成本与失败不对称性","primary_category":"cs.LG","date":"2026-09-09","score":0,"bucket":"other","tags":["模型量化","激活引导","行为控制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.06473","has_summary":false},{"id":"2609.06976","title":"HealthLoopQA: A Context-Aware Question Answering Benchmark for Interpreting Wearable Monitoring Data in Diabetes Care","zh_title":"HealthLoopQA：面向糖尿病护理中可穿戴监测数据解读的上下文感知问答基准","primary_category":"cs.LG","date":"2026-09-09","score":0,"bucket":"other","tags":["医疗问答基准","LLM推理评测","可穿戴数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.06976","has_summary":false},{"id":"2609.04243","title":"Multi-dimensional Bias in Modeling Multi-dimensional Preferences: Evaluating the Ability of Synthetic Agents to Replace Human Participants in Conjoint Experiments","zh_title":"多维偏好建模中的多维偏差：评估合成代理在联合实验中替代人类参与者的能力","primary_category":"cs.MA","date":"2026-09-07","score":10,"bucket":"selected","tags":["LLM仿真","联合实验","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.04243","has_summary":true},{"id":"2609.04485","title":"Cultural Misalignment in Large Language Models: Detection, Measurement, and Mitigation Through Targeted Fine-Tuning","zh_title":"大语言模型中的文化错位：通过定向微调进行检测、测量与缓解","primary_category":"cs.CL","date":"2026-09-07","score":9,"bucket":"selected","tags":["LLM仿真","文化偏差","价值观调查"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.04485","has_summary":true},{"id":"2609.05037","title":"How do LLMs Evaluate Perceived Moral Agency? Investigating Moral Decision-Making in Human-Artificial Agents Interactions","zh_title":"LLM如何评估感知道德能动性？探究人机交互中的道德决策","primary_category":"cs.CL","date":"2026-09-07","score":9,"bucket":"selected","tags":["LLM仿真","道德决策","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05037","has_summary":true},{"id":"2609.05009","title":"Language models judge war differently when tested for alignment","zh_title":"语言模型在对齐测试下对战争的判断不同","primary_category":"cs.AI","date":"2026-09-07","score":8,"bucket":"selected","tags":["LLM仿真","决策偏差","对齐评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05009","has_summary":true},{"id":"2609.03221","title":"Counterfactual Fairness Audits of Multi-Step Clinical LLM Agents Require a Measured Per-Action Instability Floor","zh_title":"多步临床LLM智能体的反事实公平性审计需要测量每个动作的不稳定性下限","primary_category":"cs.CL","date":"2026-09-07","score":7,"bucket":"pending","tags":["公平性审计","LLM智能体","可靠性评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.03221","has_summary":true},{"id":"2609.04373","title":"Why Better Models Can Create Riskier Systems: Evidence from LLM Agents in Financial Markets","zh_title":"为什么更好的模型会创造更危险的系统：来自金融市场中LLM智能体的证据","primary_category":"cs.AI","date":"2026-09-07","score":7,"bucket":"pending","tags":["LLM智能体","金融市场仿真","系统风险"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.04373","has_summary":true},{"id":"2609.04738","title":"Aplaud: Adaptive Personalized Low-Rank Decomposition for User-Specific LLM","zh_title":"Aplaud：面向用户特定LLM的自适应个性化低秩分解","primary_category":"cs.AI","date":"2026-09-07","score":7,"bucket":"pending","tags":["LLM个性化","调查回答预测","低秩适配"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.04738","has_summary":true},{"id":"2609.05245","title":"Do LLMs Exhibit Coherent Knowledge Structures in Mathematical Reasoning? A Perspective from Knowledge Space Theory","zh_title":"大语言模型在数学推理中是否表现出连贯的知识结构？来自知识空间理论的视角","primary_category":"cs.AI","date":"2026-09-07","score":7,"bucket":"pending","tags":["LLM知识结构","人类对照","仿真偏差"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.05245","has_summary":true},{"id":"2609.05036","title":"Moral Competence Before Moral Content: Why LLM Agents Lack the Prerequisites for Coherent Alignment","zh_title":"道德内容之前的道德能力：为何LLM智能体缺乏连贯对齐的前提条件","primary_category":"cs.AI","date":"2026-09-07","score":6,"bucket":"other","tags":["LLM道德决策","对齐评估","行为一致性"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2609.05036","has_summary":false},{"id":"2609.04866","title":"LLM-Assisted Behavioural and Scenario Augmentation for Agent-Based Energy Adoption Models","zh_title":"基于LLM的行为与场景增强用于智能体能源采纳模型","primary_category":"cs.AI","date":"2026-09-07","score":6,"bucket":"other","tags":["LLM辅助仿真","智能体建模","能源政策"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.04866","has_summary":false},{"id":"2609.05227","title":"CABAL: Multi-Agent Simulacra for Tracing the Effects of Collusive Bidding in Peer Review","zh_title":"CABAL：用于追踪同行评审中合谋投标影响的多智能体仿真框架","primary_category":"cs.AI","date":"2026-09-07","score":6,"bucket":"other","tags":["多智能体仿真","同行评审","合谋行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.05227","has_summary":false},{"id":"2609.05345","title":"Moral Advice as Interactional Negotiation: Framing, User Pressure, and Social Position in Large Language Model Responses","zh_title":"作为互动协商的道德建议：大语言模型回应中的框架、用户压力与社会位置","primary_category":"cs.CY","date":"2026-09-07","score":6,"bucket":"other","tags":["LLM道德建议","立场稳定性","人机交互"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.05345","has_summary":false},{"id":"2609.04428","title":"A Repeated-Measurement Study for Cultural Analytics of English Song Lyrics Using Five Large Language Models","zh_title":"使用五个大语言模型对英文歌词进行文化分析的重测信度研究","primary_category":"cs.LG","date":"2026-09-07","score":6,"bucket":"other","tags":["LLM标注","文化分析","测量可靠性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.04428","has_summary":false},{"id":"2608.26178","title":"AI Revealed Preferences","zh_title":"AI揭示的偏好","primary_category":"cs.AI","date":"2026-09-07","score":5,"bucket":"other","tags":["LLM偏好","AI福利","对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.26178","has_summary":false},{"id":"2609.04835","title":"On Epistemic Diversity in Large Language Models","zh_title":"论大语言模型中的认知多样性","primary_category":"cs.CL","date":"2026-09-07","score":5,"bucket":"other","tags":["LLM评估","认知多样性","知识生成"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.04835","has_summary":false},{"id":"2609.04855","title":"CC-Mediation: Evaluating Large Language Models for Cross-Cultural Conflict Mediation","zh_title":"CC-Mediation：评估大语言模型在跨文化冲突调解中的表现","primary_category":"cs.CL","date":"2026-09-07","score":5,"bucket":"other","tags":["跨文化调解","LLM评估","对话系统"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.04855","has_summary":false},{"id":"2609.05143","title":"A Human-in-the-Loop Framework for AI-Assisted Scoring in Large-Scale Writing Assessment","zh_title":"大规模写作评估中AI辅助评分的人机协同框架","primary_category":"cs.CL","date":"2026-09-07","score":5,"bucket":"other","tags":["AI辅助评分","人机协同","教育评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05143","has_summary":false},{"id":"2609.04667","title":"ERPBench: Evaluating LLM Agents for Enterprise Decision-Making Across Competitive Market Ecologies","zh_title":"ERPBench：评估跨竞争市场生态的企业决策 LLM 智能体","primary_category":"cs.AI","date":"2026-09-07","score":5,"bucket":"other","tags":["LLM智能体","企业决策模拟","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.04667","has_summary":false},{"id":"2609.05346","title":"Who Should Grade My Work? Student Perspectives on Transparent AI-Assisted Writing Assessment in Higher Education","zh_title":"谁该给我的作业评分？高等教育中透明AI辅助写作评估的学生视角","primary_category":"cs.AI","date":"2026-09-07","score":5,"bucket":"other","tags":["AI辅助评估","学生感知","评分权威"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.05346","has_summary":false},{"id":"2608.22411","title":"Don' t Box Me In: Dynamic Cultural Adaptation and Cognitive Tracking for Social Understanding","zh_title":"别把我框住：面向社会理解的动态文化适应与认知追踪","primary_category":"cs.CL","date":"2026-09-07","score":3,"bucket":"other","tags":["文化适应","对话系统","社交智能"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22411","has_summary":false},{"id":"2608.12669","title":"From Fair Representation to Just Recognition in Generative AI","zh_title":"从公平表征到生成式AI中的承认正义","primary_category":"cs.CY","date":"2026-09-07","score":2,"bucket":"other","tags":["AI伦理","公平性","政治哲学"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12669","has_summary":false},{"id":"2608.21618","title":"AI-Augmented Inquiry and Regulation in Hybrid Systems: A Control Allocation Architecture for Preserving Epistemic Agency in Hybrid Human-AI Cognition","zh_title":"混合系统中AI增强的探究与调控：一种在混合人机认知中保留认知能动性的控制分配架构","primary_category":"physics.ed-ph","date":"2026-09-07","score":2,"bucket":"other","tags":["人机协作","认知调控","教育技术"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.21618","has_summary":false},{"id":"2608.28382","title":"When Linguistic and Internal Confidence Diverge in Large Language Models","zh_title":"当大语言模型的语言置信度与内部置信度出现分歧","primary_category":"cs.CL","date":"2026-09-07","score":2,"bucket":"other","tags":["置信度校准","模型可靠性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.28382","has_summary":false},{"id":"2608.31115","title":"InsightToast: Proactive Information Retrieval & Glanceable Visualization in the Side Channel of Data-Rich Meetings","zh_title":"InsightToast：数据丰富会议侧信道中的主动信息检索与可扫视可视化","primary_category":"cs.HC","date":"2026-09-07","score":2,"bucket":"other","tags":["多智能体系统","信息检索","会议支持"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.31115","has_summary":false},{"id":"2609.04013","title":"LLM4CKD: Large Language Models for Early Stage Chronic Kidney Disease Screening","zh_title":"LLM4CKD：用于早期慢性肾病筛查的大语言模型","primary_category":"cs.AI","date":"2026-09-07","score":2,"bucket":"other","tags":["LLM医疗应用","零样本学习","疾病筛查"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04013","has_summary":false},{"id":"2609.04290","title":"Evidence Integration in Large Language Models","zh_title":"大语言模型中的证据整合","primary_category":"cs.CL","date":"2026-09-07","score":2,"bucket":"other","tags":["LLM推理","证据整合","模型行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04290","has_summary":false},{"id":"2609.04384","title":"You Really Didn't Get That? Benchmarking Social Pragmatic Inference for Indirect and Playful Chinese Online Comments","zh_title":"你真的没听懂吗？面向中文网络评论间接与戏谑表达的社会语用推理基准测试","primary_category":"cs.CL","date":"2026-09-07","score":2,"bucket":"other","tags":["语用推理","基准测试","社交媒体"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04384","has_summary":false},{"id":"2609.04463","title":"Shared circuits predict whether LLMs generalize across formats in arithmetic reasoning","zh_title":"共享电路预测大语言模型在算术推理中跨格式泛化的能力","primary_category":"cs.CL","date":"2026-09-07","score":2,"bucket":"other","tags":["LLM推理","电路分析","跨格式泛化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04463","has_summary":false},{"id":"2609.04484","title":"Patterns of Priming in Production: Lexical, Semantic and Structural Alignment in Language Model Generation","zh_title":"生成中的启动模式：语言模型生成中的词汇、语义和结构对齐","primary_category":"cs.CL","date":"2026-09-07","score":2,"bucket":"other","tags":["语言模型","结构启动","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04484","has_summary":false},{"id":"2609.04206","title":"Auditing Bias and Safety in Voice AI Customer Care","zh_title":"语音AI客服中的偏见与安全审计","primary_category":"eess.AS","date":"2026-09-07","score":2,"bucket":"other","tags":["语音AI","公平性审计","客服系统"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.04206","has_summary":false},{"id":"2609.04303","title":"Abstraction Agent","zh_title":"抽象智能体","primary_category":"cs.MA","date":"2026-09-07","score":2,"bucket":"other","tags":["游戏AI","LLM应用","知识提取"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.04303","has_summary":false},{"id":"2609.04850","title":"ElderBench: Benchmarking Autonomous Mobile Agents for Older Adults","zh_title":"ElderBench：面向老年人的自主移动代理基准测试","primary_category":"cs.AI","date":"2026-09-07","score":2,"bucket":"other","tags":["GUI代理","老年人","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.04850","has_summary":false},{"id":"2609.04894","title":"From Language Models to World-Acting Systems: Progress and Limits of Agentic AI across Digital, Social, Virtual, and Physical Environments","zh_title":"从语言模型到世界行动系统：智能体AI在数字、社会、虚拟和物理环境中的进展与局限","primary_category":"cs.AI","date":"2026-09-07","score":2,"bucket":"other","tags":["智能体系统","自主性","系统集成"],"rubric_hits":["C1","C2"],"abs_url":"https://arxiv.org/abs/2609.04894","has_summary":false},{"id":"2609.05363","title":"Distill Globally, Adapt Locally: Reasoning Distillation and Product-Type Test-Time Training for Scalable Trade-Up Recommendation","zh_title":"全局蒸馏，局部适应：面向可扩展升级推荐的理由蒸馏与产品类型测试时训练","primary_category":"cs.LG","date":"2026-09-07","score":2,"bucket":"other","tags":["推荐系统","知识蒸馏","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05363","has_summary":false},{"id":"2608.06621","title":"NxN E-valuation: Hypothesis Certification via a Conformal CRT Null","zh_title":"NxN E-valuation：通过共形CRT零假设进行假设认证","primary_category":"cs.AI","date":"2026-09-07","score":0,"bucket":"other","tags":["假设检验","LLM验证","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06621","has_summary":false},{"id":"2609.04556","title":"Rhythms of Work: Multi-Scale Interpretation of Human Behavioral Traces for Workplace Agents","zh_title":"工作节奏：面向工作场所智能体的人类行为轨迹多尺度解释","primary_category":"cs.CL","date":"2026-09-07","score":0,"bucket":"other","tags":["行为轨迹分析","多尺度解释","工作场所智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04556","has_summary":false},{"id":"2609.04841","title":"MABPD: Multi-Agent Bias Probing & Detection via Structured Argument Debate","zh_title":"MABPD：通过结构化论证辩论进行多智能体偏见探测与检测","primary_category":"cs.CL","date":"2026-09-07","score":0,"bucket":"other","tags":["多智能体系统","偏见检测","NLP应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04841","has_summary":false},{"id":"2609.04272","title":"Evaluating Large Language Models for Forced Outage Risk Prediction: Benefits and Comparison to Machine Learning","zh_title":"评估大语言模型在强制停电风险预测中的表现：优势及与机器学习的比较","primary_category":"cs.LG","date":"2026-09-07","score":0,"bucket":"other","tags":["LLM预测","停电风险","零样本分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04272","has_summary":false},{"id":"2609.04445","title":"Conformity Breaks Conformal Prediction","zh_title":"从众打破共形预测","primary_category":"cs.LG","date":"2026-09-07","score":0,"bucket":"other","tags":["多智能体系统","共形预测","LLM可靠性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04445","has_summary":false},{"id":"2609.04286","title":"From Matching Models to Recruiting Agents: A Systematized Narrative Review of AI Recruitment Systems, Evaluation, and Governance","zh_title":"从匹配模型到招聘智能体：AI招聘系统、评估与治理的系统化叙述性综述","primary_category":"cs.AI","date":"2026-09-07","score":0,"bucket":"other","tags":["AI招聘","多智能体系统","系统综述"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04286","has_summary":false},{"id":"2609.04880","title":"Reinforcement Learning for Sequential Solar PV Policy Design under Uncertainty: An Agent-Based Approach","zh_title":"不确定性下太阳能光伏政策序列设计的强化学习：基于智能体的方法","primary_category":"cs.AI","date":"2026-09-07","score":0,"bucket":"other","tags":["强化学习","智能体建模","政策设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04880","has_summary":false},{"id":"2609.04542","title":"Matched Starts, Divergent Objects: How Human-AI Collaboration Forms What It Explains","zh_title":"匹配起点，分歧对象：人机协作如何形成其解释内容","primary_category":"cs.HC","date":"2026-09-07","score":0,"bucket":"other","tags":["人机协作","学术知识生产","过程分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04542","has_summary":false},{"id":"2609.05115","title":"Creators Have Difficulty Abandoning Ideas They Generated","zh_title":"创造者难以放弃自己产生的想法","primary_category":"econ.GN","date":"2026-09-07","score":0,"bucket":"other","tags":["创造力","决策偏差","人类实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.05115","has_summary":false},{"id":"2609.03215","title":"SWIM: Student Writing Simulation via Proficiency-Conditioned Generation","zh_title":"SWIM：基于熟练度条件生成的学生写作仿真","primary_category":"cs.CL","date":"2026-09-04","score":9,"bucket":"selected","tags":["LLM仿真","学生写作","熟练度对齐"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.03215","has_summary":true},{"id":"2609.03553","title":"GPS-Bench: A Governance Policy Benchmark for Automating Policy Analysis","zh_title":"GPS-Bench：用于自动化政策分析的治理政策基准","primary_category":"cs.AI","date":"2026-09-04","score":8,"bucket":"selected","tags":["LLM仿真","政策模拟","基准测试"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.03553","has_summary":true},{"id":"2609.03218","title":"The Analyst in the Prompt: Role, Retrieval, and Memory Biases in LLM Financial Analysis","zh_title":"提示中的分析师：LLM金融分析中的角色、检索与记忆偏差","primary_category":"cs.CL","date":"2026-09-04","score":7,"bucket":"pending","tags":["LLM偏差","金融分析","个性化影响"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.03218","has_summary":true},{"id":"2609.04198","title":"Clean Engineering, Unstable Measurement: A Preregistered Reliability Failure of Black-Box LLM Observers on Shared Endpoints","zh_title":"清洁工程，不稳定测量：黑盒LLM观察者在共享端点上的预注册可靠性失败","primary_category":"cs.AI","date":"2026-09-04","score":7,"bucket":"pending","tags":["LLM可靠性","测量工具","预注册"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.04198","has_summary":true},{"id":"2609.04047","title":"The Dice Roll Method: A Standardized Protocol for Repeated-Query Auditing of Large Language Model Brand Recommendations","zh_title":"骰子滚动法：大语言模型品牌推荐重复查询审计的标准化协议","primary_category":"cs.IR","date":"2026-09-04","score":5,"bucket":"other","tags":["LLM审计","品牌推荐","稳定性协议"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.04047","has_summary":false},{"id":"2609.04127","title":"Epistemic Warrant for LLM Recommendations: Characterizing the Basis for Reliance When Ground Truth Is Unavailable","zh_title":"LLM推荐的认识论保证：在无真值情况下表征依赖基础","primary_category":"cs.AI","date":"2026-09-04","score":5,"bucket":"other","tags":["LLM推荐","认识论保证","决策支持"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.04127","has_summary":false},{"id":"2609.03507","title":"LongCounsel-8: A Benchmark Suite for Longitudinal Depression Tracking from Multi-Session Counseling Dialogues","zh_title":"LongCounsel-8：多会话咨询对话纵向抑郁追踪基准套件","primary_category":"cs.LG","date":"2026-09-04","score":5,"bucket":"other","tags":["LLM仿真","心理健康","基准数据集"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.03507","has_summary":false},{"id":"2609.03344","title":"Large-Language Models as a Cognitive Virus","zh_title":"大语言模型作为认知病毒","primary_category":"physics.soc-ph","date":"2026-09-04","score":5,"bucket":"other","tags":["社会模拟","技术扩散","临界相变"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.03344","has_summary":false},{"id":"2609.02992","title":"Tempting the Agent: The Economics of Reputation without Persistent Identity in AI Agent Markets","zh_title":"诱惑智能体：AI代理市场中无持久身份下的声誉经济学","primary_category":"q-fin.GN","date":"2026-09-04","score":5,"bucket":"other","tags":["AI代理市场","声誉机制","经济模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.02992","has_summary":false},{"id":"2609.02719","title":"Large Language Model-Driven Context-Aware Eco-Feedback Generation and Evaluation","zh_title":"大语言模型驱动的上下文感知生态反馈生成与评估","primary_category":"cs.HC","date":"2026-09-04","score":2,"bucket":"other","tags":["LLM应用","节能反馈","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.02719","has_summary":false},{"id":"2609.02890","title":"Bounded Personas Match Retrieval on Classification but Not Regression for a Frozen Agent","zh_title":"有界人格在分类任务上与检索匹配但在回归任务上不匹配：以冻结代理为例","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["个性化语言代理","人格蒸馏","检索增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02890","has_summary":false},{"id":"2609.02895","title":"BharatGather: A Culturally-Informed Benchmark Dataset for Misinformation and Fake News Detection in Indian Public Events","zh_title":"BharatGather：面向印度公共事件的具有文化信息的错误信息和假新闻检测基准数据集","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["假新闻检测","数据集","LLM数据增强"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02895","has_summary":false},{"id":"2609.03370","title":"FrameBench:A Language Understanding Benchmark Based on Frame Semantics","zh_title":"FrameBench：基于框架语义的语言理解基准","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["NLP评测","框架语义","语言理解"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03370","has_summary":false},{"id":"2609.03394","title":"Chiaroscuro for Emotions: A Contrastive Emotion Benchmark Grounded in Appraisal Theory","zh_title":"情感明暗对比：基于评价理论的对比情感基准","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["情感识别","基准测试","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03394","has_summary":false},{"id":"2609.03577","title":"Language, Language Models, and What We're Talking About","zh_title":"语言、语言模型与我们谈论的内容","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["语言模型","语言本质","NLP批判"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03577","has_summary":false},{"id":"2609.03652","title":"The Impact of Synthetic Data Augmentation on Discourse-Pragmatic Function Classification","zh_title":"合成数据增强对语篇-语用功能分类的影响","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["合成数据增强","语篇功能分类","低资源NLP"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03652","has_summary":false},{"id":"2609.03687","title":"A Circuit for Plural Reference: How LLMs Represent and Retrieve Singular and Plural Entities","zh_title":"复数指代电路：LLM如何表示和检索单数与复数实体","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["机制可解释性","指代消解","注意力分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03687","has_summary":false},{"id":"2609.03967","title":"Investigating the Ability of Large Language Models to Analyze Recipes for Diabetes","zh_title":"探究大语言模型分析糖尿病食谱的能力","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["LLM能力评测","糖尿病食谱","提示工程"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03967","has_summary":false},{"id":"2609.04022","title":"Representational alignment yields generalizable safety in language models","zh_title":"表征对齐在语言模型中产生可泛化的安全性","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["安全对齐","表征学习","道德判断"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04022","has_summary":false},{"id":"2609.04048","title":"Translation as a Decision Space: A Multi-Agent Perspective on Low-Resource Dialect Generation","zh_title":"翻译作为决策空间：低资源方言生成的多智能体视角","primary_category":"cs.CL","date":"2026-09-04","score":2,"bucket":"other","tags":["多智能体系统","机器翻译","低资源方言"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04048","has_summary":false},{"id":"2609.03923","title":"Speak for Me: Giving LLMs the Situational Awareness to Participate in a Meeting","zh_title":"替我发言：赋予大语言模型参与会议的情境意识","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["多智能体系统","会议代理","对话生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03923","has_summary":false},{"id":"2609.03402","title":"A Prompt-Engineering Approach to Develop Scalable, Flexible, and Real-Time Hybrid Micro-Level Personalization in a General Purpose AI Teaching Assistant","zh_title":"一种在通用AI教学助手中开发可扩展、灵活、实时混合微观个性化的提示工程方法","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["AI教学助手","个性化","提示工程"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.03402","has_summary":false},{"id":"2609.03407","title":"Caught in the Story: Narrative Captivity in Multi-turn LLMs Conversation","zh_title":"陷入叙事：多轮LLM对话中的叙事俘获","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["LLM道德判断","多轮对话","模型偏差"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.03407","has_summary":false},{"id":"2609.03920","title":"Value-Preserving Architectures for Agentic AI Systems","zh_title":"面向智能体AI系统的价值保持架构","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["多智能体系统","价值对齐","软件架构"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03920","has_summary":false},{"id":"2609.04141","title":"Efficient Test-Time Adaptation through Human-AI Interaction","zh_title":"通过人机交互实现高效测试时适应","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["人机交互","个性化适应","AI代理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04141","has_summary":false},{"id":"2609.04166","title":"From Deceptive Outputs to Deceptive Mechanisms: A Causal Framework for Language-Model Deception Research","zh_title":"从欺骗性输出到欺骗性机制：语言模型欺骗研究的因果框架","primary_category":"cs.AI","date":"2026-09-04","score":2,"bucket":"other","tags":["LLM欺骗","因果框架","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04166","has_summary":false},{"id":"2609.03192","title":"Where Reliability Lives: Experimental Localisation of Behavioural Properties in an Agent System","zh_title":"可靠性所在：智能体系统中行为属性的实验定位","primary_category":"cs.MA","date":"2026-09-04","score":2,"bucket":"other","tags":["多智能体系统","行为属性","实验方法"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03192","has_summary":false},{"id":"2608.13775","title":"Structured Payment in Pawnshop Borrowing: Mandates vs. Choice","zh_title":"典当借款中的结构化还款：强制与选择","primary_category":"econ.GN","date":"2026-09-04","score":0,"bucket":"other","tags":["典当贷款","随机对照试验","还款结构"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.13775","has_summary":false},{"id":"2608.22793","title":"TRACE: A Self-Evolving Skill Bank for Consistent, Limit-Aware LLM Agents","zh_title":"TRACE：用于一致、边界感知LLM智能体的自进化技能库","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["LLM智能体","可靠性","技能库"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22793","has_summary":false},{"id":"2608.21601","title":"K-Bench: measuring model performance on real scientific agent requests","zh_title":"K-Bench：衡量模型在真实科学agent请求上的表现","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["科学agent评测","基准测试","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.21601","has_summary":false},{"id":"2609.02899","title":"Contamination Inflates Scores but Rarely Reorders Large Language Model Leaderboards","zh_title":"污染推高分数但很少改变大语言模型排行榜顺序","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["基准污染","排行榜","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02899","has_summary":false},{"id":"2609.02942","title":"Judging LLM-as-a-Judge: Concerning Rubric Artifacts in LLM-based Automated Text Generation Evaluation","zh_title":"评判LLM作为评判者：关于基于LLM的自动文本生成评估中的评分标准伪影","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["LLM评估","自动评分","可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02942","has_summary":false},{"id":"2609.03160","title":"No country for old linguists: LLM-brain alignment underdetermines neural computation","zh_title":"老语言学家无立足之地：LLM-大脑对齐不足以确定神经计算","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["LLM-大脑对齐","计算神经科学","哲学批判"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03160","has_summary":false},{"id":"2609.03213","title":"LLMs Learn Better In-Context from Rules than from Examples","zh_title":"大语言模型从规则中比从示例中更好地进行上下文学习","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["上下文学习","指令跟随","少样本提示"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03213","has_summary":false},{"id":"2609.04194","title":"Legibility is Not Interpretability: Comparing Judged and Actual Importance in Chain-Of-Thought Reasoning","zh_title":"可读性不等于可解释性：比较思维链推理中判断重要性与实际重要性","primary_category":"cs.CL","date":"2026-09-04","score":0,"bucket":"other","tags":["思维链","可解释性","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.04194","has_summary":false},{"id":"2609.02959","title":"The Geometry of Ignorance: LLMs Know When to Temper Bayesian Priors","zh_title":"无知几何：大语言模型知道何时调节贝叶斯先验","primary_category":"cs.LG","date":"2026-09-04","score":0,"bucket":"other","tags":["模型可解释性","贝叶斯推断","语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02959","has_summary":false},{"id":"2609.03460","title":"Beyond \"Made with AI\": Visualizing Provenance Density to Mitigate the Transparency Penalty","zh_title":"超越“AI制造”：可视化来源密度以缓解透明度惩罚","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["AI透明度","用户研究","信息可视化"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.03460","has_summary":false},{"id":"2609.03588","title":"KC-Bench: A Dynamic Interactive Benchmark for Evaluating Knowledge Conflicts in LLM Agents","zh_title":"KC-Bench：评估LLM智能体知识冲突的动态交互基准","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["LLM智能体","知识冲突","基准评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03588","has_summary":false},{"id":"2609.03635","title":"Analysis of Prompt Engineering for Drug Toxicity Prediction","zh_title":"药物毒性预测的提示工程分析","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["药物毒性预测","提示工程","机器学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.03635","has_summary":false},{"id":"2609.03860","title":"Adapting to Evolving Requirements: Agentic AI for Retail Supply Chain Operations","zh_title":"适应不断变化的需求：面向零售供应链运营的智能体AI","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["多智能体系统","供应链优化","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03860","has_summary":false},{"id":"2609.04170","title":"A Case Study on Emergent Cheating and Whistleblowing in Autonomous Research Swarms","zh_title":"自主研究群体中涌现的作弊与举报行为案例研究","primary_category":"cs.AI","date":"2026-09-04","score":0,"bucket":"other","tags":["多智能体系统","涌现行为","知识共享治理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.04170","has_summary":false},{"id":"2609.03425","title":"The Civilization Framework: Sovereign-Anchored Communication Between Personal Multi-Agent Systems","zh_title":"文明框架：个人多智能体系统之间的主权锚定通信","primary_category":"cs.MA","date":"2026-09-04","score":0,"bucket":"other","tags":["多智能体系统","通信协议","AI基础设施"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03425","has_summary":false},{"id":"2609.03853","title":"Bridging Formal and Perceived Fairness: Development of an Interdisciplinary Framework in Algorithmic Decision-Making","zh_title":"桥接形式公平与感知公平：算法决策中跨学科框架的发展","primary_category":"cs.CY","date":"2026-09-04","score":0,"bucket":"other","tags":["算法公平","人机交互","跨学科框架"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2609.03853","has_summary":false},{"id":"2609.03422","title":"Inferred Generative-Process Diversity Predicts Correlated Failure Across Language Models","zh_title":"推断的生成过程多样性预测语言模型间的相关失败","primary_category":"cs.LG","date":"2026-09-04","score":0,"bucket":"other","tags":["多模型系统","多样性度量","失败相关性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03422","has_summary":false},{"id":"2609.03177","title":"Frontier LLMs are effective batch optimizers: Assessing reasoning models in continuous and discrete settings","zh_title":"前沿大语言模型是有效的批量优化器：评估连续和离散设置中的推理模型","primary_category":"cs.LG","date":"2026-09-04","score":0,"bucket":"other","tags":["LLM优化器","批量优化","推理模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.03177","has_summary":false},{"id":"2609.02526","title":"When Persona Attributes Improve Population Alignment in Large Language Models","zh_title":"当人物属性改善大语言模型中的群体对齐时","primary_category":"cs.CL","date":"2026-09-03","score":10,"bucket":"selected","tags":["LLM仿真","调查预测","人物提示"],"rubric_hits":["A1","A2","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2609.02526","has_summary":true},{"id":"2609.02580","title":"Competitive Market Behavior of LLMs","zh_title":"大语言模型的竞争性市场行为","primary_category":"cs.MA","date":"2026-09-03","score":10,"bucket":"selected","tags":["LLM仿真","经济学实验","市场机制"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.02580","has_summary":true},{"id":"2601.22396","title":"Culturally Grounded Personas in Large Language Models: Characterization and Alignment with Socio-Psychological Value Frameworks","zh_title":"大语言模型中文化扎根的人格：表征及与社会心理价值框架的对齐","primary_category":"cs.CL","date":"2026-09-03","score":9,"bucket":"selected","tags":["LLM仿真","文化价值观","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2601.22396","has_summary":true},{"id":"2609.02122","title":"AI agents reshape consensus formation in human groups","zh_title":"AI智能体重塑人类群体中的共识形成","primary_category":"cs.CL","date":"2026-09-03","score":9,"bucket":"selected","tags":["LLM仿真","人机交互","共识形成"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.02122","has_summary":true},{"id":"2609.01902","title":"Accurate in space, unreliable in time: how LLMs represent national cultural change","zh_title":"空间准确，时间不可靠：大语言模型如何表征国家文化变迁","primary_category":"cs.CY","date":"2026-09-03","score":9,"bucket":"selected","tags":["文化仿真","算法保真度","时间偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.01902","has_summary":true},{"id":"2609.02512","title":"Beauty is in the AI of the beholder: MLLMs systematically overrate facial attractiveness","zh_title":"美在AI眼中：多模态大模型系统性高估面部吸引力","primary_category":"cs.CV","date":"2026-09-03","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","偏差评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.02512","has_summary":true},{"id":"2609.01867","title":"Thinking effort aligns between humans and reasoning models in abductive reasoning","zh_title":"溯因推理中人类与推理模型的思维努力对齐","primary_category":"cs.CL","date":"2026-09-03","score":8,"bucket":"selected","tags":["LLM仿真","认知对齐","溯因推理"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.01867","has_summary":true},{"id":"2609.02277","title":"Auditory Illusion Benchmark for Large Audio Language Models","zh_title":"大型音频语言模型的听觉错觉基准","primary_category":"cs.SD","date":"2026-09-03","score":8,"bucket":"selected","tags":["听觉错觉","人类感知仿真","模型评估"],"rubric_hits":["A1","A2","B1"],"abs_url":"https://arxiv.org/abs/2609.02277","has_summary":true},{"id":"2608.27309","title":"Difference-in-Differences on a Censored Rating Scale Can Manufacture an Effect: Evidence from a Pre-Registered LLM-Judge Audit","zh_title":"截断评分量表上的双重差分可能制造效应：来自预注册LLM法官审计的证据","primary_category":"cs.CL","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM法官","偏差审计","方法论批判"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.27309","has_summary":true},{"id":"2608.27463","title":"Rating the Raters: Rasch Measurement Theory for LLM Evaluation","zh_title":"评估评分者：用于LLM评估的Rasch测量理论","primary_category":"cs.AI","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM评估","测量理论","评分者偏差"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.27463","has_summary":true},{"id":"2609.01794","title":"Disentangling Statistical Preemption from Entrenchment in Language Models' Avoidance of Overgeneralization","zh_title":"在语言模型避免过度泛化中区分统计抢占与固化","primary_category":"cs.CL","date":"2026-09-03","score":7,"bucket":"pending","tags":["语言习得","认知建模","人类对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.01794","has_summary":true},{"id":"2609.01918","title":"Grounded, Compute-Efficient LLM Policy Agents for Energy-Poverty Equity in Physically-Constrained Peer-to-Peer Energy Markets","zh_title":"物理约束点对点能源市场中面向能源贫困公平的接地气、计算高效LLM策略智能体","primary_category":"cs.CL","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM仿真","能源市场","公平性评估"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.01918","has_summary":true},{"id":"2609.02707","title":"Door-in-the-Face Requests and Refusal Behaviour in Large Language Models","zh_title":"大语言模型中的登门槛请求与拒绝行为","primary_category":"cs.AI","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM行为实验","说服技巧","模型对比"],"rubric_hits":["A1","B4"],"abs_url":"https://arxiv.org/abs/2609.02707","has_summary":true},{"id":"2609.02797","title":"Dutch Books for Language Models","zh_title":"语言模型的荷兰赌：概率预测的连贯性评估","primary_category":"econ.GN","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM概率预测","连贯性评估","决策可靠性"],"rubric_hits":["A2","B3"],"abs_url":"https://arxiv.org/abs/2609.02797","has_summary":true},{"id":"2609.01815","title":"Induction and Inquiry via Probabilistic Reasoning over Language and Code","zh_title":"通过语言与代码的概率推理进行归纳与探究","primary_category":"cs.AI","date":"2026-09-03","score":7,"bucket":"pending","tags":["认知建模","贝叶斯学习","人类行为复现"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.01815","has_summary":true},{"id":"2609.02821","title":"AI Contextual Measurement for Recovering Individual and Group-Level Effects: Validation Against Survey Measures and an Occupational Application","zh_title":"AI情境测量用于恢复个体与群体效应：基于调查测量的验证及职业应用","primary_category":"cs.AI","date":"2026-09-03","score":7,"bucket":"pending","tags":["AI测量","验证框架","社会调查"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.02821","has_summary":true},{"id":"2609.01627","title":"The Utility of LLMs in Recommender Systems Explanation Evaluation","zh_title":"大语言模型在推荐系统解释评估中的效用","primary_category":"cs.IR","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM评估","推荐系统解释","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2609.01627","has_summary":true},{"id":"2609.02677","title":"Eliciting ESG Preferences for Reinforcement Learning-Based Portfolio Optimization","zh_title":"基于强化学习的投资组合优化中ESG偏好的获取","primary_category":"q-fin.PM","date":"2026-09-03","score":7,"bucket":"pending","tags":["LLM仿真","投资组合优化","偏好获取"],"rubric_hits":["A1","A3","B2"],"abs_url":"https://arxiv.org/abs/2609.02677","has_summary":true},{"id":"2609.02092","title":"Beyond Outcome Gaps: Process-Aware Fairness Diagnosis for LLM-based Multi-Agent Decision Systems","zh_title":"超越结果差距：基于LLM的多智能体决策系统的过程感知公平性诊断","primary_category":"cs.AI","date":"2026-09-03","score":6,"bucket":"other","tags":["LLM多智能体","公平性诊断","招聘模拟"],"rubric_hits":["D3","A3"],"abs_url":"https://arxiv.org/abs/2609.02092","has_summary":false},{"id":"2609.02620","title":"Collective creativity in hybrid societies","zh_title":"混合社会中的集体创造力","primary_category":"cs.AI","date":"2026-09-03","score":6,"bucket":"other","tags":["LLM社会模拟","混合集体","创造力"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.02620","has_summary":false},{"id":"2608.21377","title":"Agentic Scaffolding Amplifies Sycophantic Behavior in Large Language Models","zh_title":"智能体脚手架放大大型语言模型中的谄媚行为","primary_category":"cs.CL","date":"2026-09-03","score":5,"bucket":"other","tags":["LLM行为","谄媚偏差","智能体系统"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.21377","has_summary":false},{"id":"2609.02054","title":"A Tri-Agent Framework for Evaluating and Aligning Question Clarification Capabilities of Large Language Models","zh_title":"用于评估和对齐大语言模型问题澄清能力的三智能体框架","primary_category":"cs.CL","date":"2026-09-03","score":5,"bucket":"other","tags":["LLM评估","多智能体","对话系统"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.02054","has_summary":false},{"id":"2609.02322","title":"What Is Worth Representing? Representational Empowerment for Continual Model Construction","zh_title":"什么值得表征？持续模型构建中的表征赋权","primary_category":"cs.LG","date":"2026-09-03","score":5,"bucket":"other","tags":["表征学习","持续学习","LLM增强规划"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.02322","has_summary":false},{"id":"2608.12062","title":"Preference Tree Optimization: Enhancing Goal-Oriented Dialogue with Look-Ahead Simulations","zh_title":"偏好树优化：通过前瞻模拟增强目标导向对话","primary_category":"cs.CL","date":"2026-09-03","score":3,"bucket":"other","tags":["对话系统","动机访谈","偏好优化"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12062","has_summary":false},{"id":"2609.02191","title":"Examining the Vulnerability of Multi-Agent Medical Systems to Human Interventions for Clinical Reasoning","zh_title":"多智能体医疗系统对人类干预的脆弱性研究：临床推理视角","primary_category":"cs.AI","date":"2026-09-03","score":3,"bucket":"other","tags":["多智能体系统","医疗AI","诊断准确性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02191","has_summary":false},{"id":"2608.22639","title":"Poetic Heritage for Culturally Grounded Emotional Support: An Interaction Design Framework and Its Multimodal Agentic Instantiation","zh_title":"基于文化根基的情感支持的诗意遗产：交互设计框架及其多模态智能体实例","primary_category":"cs.HC","date":"2026-09-03","score":2,"bucket":"other","tags":["情感支持","多智能体系统","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22639","has_summary":false},{"id":"2608.30188","title":"GPAgentBench-2K: Benchmarking Large Language Model Agents in Complex Clinical Action Space","zh_title":"GPAgentBench-2K：在复杂临床动作空间中基准测试大语言模型智能体","primary_category":"cs.CL","date":"2026-09-03","score":2,"bucket":"other","tags":["LLM智能体","临床决策","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.30188","has_summary":false},{"id":"2609.01832","title":"Interpretable Symptom Vectors for Depression in a Large Language Model","zh_title":"大语言模型中抑郁症的可解释症状向量","primary_category":"cs.CL","date":"2026-09-03","score":2,"bucket":"other","tags":["可解释性","心理健康","表征分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.01832","has_summary":false},{"id":"2609.02275","title":"Do Large Language Models Capture the Diversity in their Training Data?","zh_title":"大语言模型是否捕捉到其训练数据中的多样性？","primary_category":"cs.CL","date":"2026-09-03","score":2,"bucket":"other","tags":["生成模型","信息论","多样性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02275","has_summary":false},{"id":"2609.02496","title":"Debias-SparseGPT: Bias-Aware Pruning for Large Language Models","zh_title":"Debias-SparseGPT：面向大语言模型的偏差感知剪枝","primary_category":"cs.CL","date":"2026-09-03","score":2,"bucket":"other","tags":["模型压缩","偏差缓解","剪枝"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02496","has_summary":false},{"id":"2609.02651","title":"WinoQueer-NL: Assessing Bias in Dutch Language Models toward LGBTQ+ Identities","zh_title":"WinoQueer-NL：评估荷兰语语言模型对LGBTQ+身份偏见","primary_category":"cs.CL","date":"2026-09-03","score":2,"bucket":"other","tags":["偏见评估","语言模型","LGBTQ+"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02651","has_summary":false},{"id":"2609.01873","title":"Epistemic Sybil Resistance: Multiplying AI Agents Without Multiplying Evidence","zh_title":"认知女巫抵抗：在不增加证据的情况下增加AI智能体","primary_category":"cs.AI","date":"2026-09-03","score":2,"bucket":"other","tags":["多智能体系统","证据聚合","推理校准"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01873","has_summary":false},{"id":"2609.02231","title":"PhoenixNest-Video: Evidence-Grounded Multimodal Agent Framework for Automated Video Interview Assessment","zh_title":"PhoenixNest-Video：基于证据的多模态智能体框架用于自动化视频面试评估","primary_category":"cs.AI","date":"2026-09-03","score":2,"bucket":"other","tags":["多模态智能体","面试评估","强化学习"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.02231","has_summary":false},{"id":"2607.24780","title":"LivingArena: Do LLMs Know What Other LLMs Don't? Peer-Probing as Scalable Evaluation","zh_title":"LivingArena：大语言模型是否知道其他模型不知道什么？基于同伴探测的可扩展评估","primary_category":"cs.AI","date":"2026-09-03","score":0,"bucket":"other","tags":["LLM评测","多智能体","知识边界探测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24780","has_summary":false},{"id":"2608.20574","title":"FlavourBench: Executable Culinary Reward Maps for Language Model Evaluation and Post-Training","zh_title":"FlavourBench：用于语言模型评估和后训练的可执行烹饪奖励地图","primary_category":"cs.AI","date":"2026-09-03","score":0,"bucket":"other","tags":["LLM评估","烹饪任务","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.20574","has_summary":false},{"id":"2608.21969","title":"ToSCA: Leveraging Hierarchical Reinforcement Learning on Temporal and Strategic Abstractions of Conversational Agents","zh_title":"ToSCA：利用对话智能体的时间和策略抽象的分层强化学习","primary_category":"cs.CL","date":"2026-09-03","score":0,"bucket":"other","tags":["对话系统","分层强化学习","自然语言生成"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21969","has_summary":false},{"id":"2609.02730","title":"CORAL: An LLM-Native Harness for Production Recommender Systems","zh_title":"CORAL：面向生产推荐系统的 LLM 原生自动化框架","primary_category":"cs.CL","date":"2026-09-03","score":0,"bucket":"other","tags":["推荐系统","LLM agent","系统优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02730","has_summary":false},{"id":"2609.01625","title":"Whose Judgments Count? Representation Gaps in Crowdsourced Content Moderation Produce Unequal Protection from Perceived Toxicity","zh_title":"谁的判断算数？众包内容审核中的代表性差距导致感知毒性保护不平等","primary_category":"cs.SI","date":"2026-09-03","score":0,"bucket":"other","tags":["内容审核","众包","代表性差距"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01625","has_summary":false},{"id":"2609.02745","title":"Incremental Pooled LLM Evaluation for Cost-Effective Retrieval Model Selection","zh_title":"用于成本效益检索模型选择的增量池化LLM评估","primary_category":"cs.IR","date":"2026-09-03","score":0,"bucket":"other","tags":["信息检索","LLM评估","RAG系统"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.02745","has_summary":false},{"id":"2609.02242","title":"Propose to Learn, Learn to Propose: Evaluability-Aware Assistance under Bounded Rationality","zh_title":"提出以学习，学习以提出：有限理性下的可评估性感知辅助","primary_category":"cs.AI","date":"2026-09-03","score":0,"bucket":"other","tags":["AI辅助","有限理性","规划"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02242","has_summary":false},{"id":"2609.02246","title":"LLM-as-a-Judge Is Not an Oracle: Why Self-Improving Agents Need Deterministic Guardrails","zh_title":"LLM作为评判者并非神谕：为何自改进智能体需要确定性护栏","primary_category":"cs.AI","date":"2026-09-03","score":0,"bucket":"other","tags":["LLM评判器","自改进智能体","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02246","has_summary":false},{"id":"2609.02750","title":"Bilevel Coordinated Reflection: A Game-Theoretic Approach to Multi-Agent LLM Systems","zh_title":"双层协调反思：多智能体LLM系统的博弈论方法","primary_category":"cs.AI","date":"2026-09-03","score":0,"bucket":"other","tags":["多智能体系统","博弈论","任务协调"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02750","has_summary":false},{"id":"2609.01976","title":"Knowing Is Not Enough: Information Retrievability as a Precondition to Effective LLM Oversight","zh_title":"知道还不够：信息可检索性作为有效LLM监督的前提","primary_category":"cs.HC","date":"2026-09-03","score":0,"bucket":"other","tags":["人机交互","LLM监督","信息检索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01976","has_summary":false},{"id":"2609.02296","title":"Meeting the Coming Wave: The Emerging Politics of AI and Work across 33 Parliaments","zh_title":"迎接浪潮：33个议会中人工智能与工作的新兴政治","primary_category":"cs.CY","date":"2026-09-03","score":0,"bucket":"other","tags":["政治学","文本分析","AI政策"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.02296","has_summary":false},{"id":"2609.01639","title":"Inverse planning of social interactions in relationships","zh_title":"人际关系中社会互动的逆向规划","primary_category":"physics.soc-ph","date":"2026-09-03","score":0,"bucket":"other","tags":["社会认知","逆向规划","人类实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01639","has_summary":false},{"id":"2607.10628","title":"Anamnesis: An Open-Source Platform for Large-Scale Backstory-Conditioned Survey Simulation","zh_title":"Anamnesis：大规模背景条件调查仿真的开源平台","primary_category":"cs.CL","date":"2026-09-02","score":10,"bucket":"selected","tags":["LLM仿真","调查模拟","人类数据对照"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.10628","has_summary":true},{"id":"2609.00222","title":"LLM-as-a-Demographic: Whom Sociodemographic Prompting Helps, and Whom It Hurts","zh_title":"LLM作为人口群体：社会人口学提示对谁有益，对谁有害","primary_category":"cs.CL","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","人口学提示","算法偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00222","has_summary":true},{"id":"2609.01591","title":"StudentSim: Training LLM-based Student Simulators","zh_title":"StudentSim：训练基于LLM的学生模拟器","primary_category":"cs.CL","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","学生模拟","行为保真度"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.01591","has_summary":true},{"id":"2609.01038","title":"Data-Driven Persona-Conditioned Agents for A/B Test Simulation","zh_title":"基于数据驱动人物画像的智能体用于A/B测试模拟","primary_category":"cs.AI","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","A/B测试","人物画像"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.01038","has_summary":true},{"id":"2609.01257","title":"Measuring the Behavioral Fidelity of Long-Horizon Human Activity Simulations","zh_title":"衡量长时程人类活动模拟的行为保真度","primary_category":"cs.AI","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","行为保真度","人类活动模拟"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.01257","has_summary":true},{"id":"2609.01275","title":"The Constitutional Coverage Trilemma in AI Governance","zh_title":"AI治理中的宪法覆盖三难困境","primary_category":"cs.LG","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","价值观对齐","人类对照"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2609.01275","has_summary":true},{"id":"2609.00009","title":"Toward a social psychology of AI: language-model agents reproduce human-like minimal-group bias","zh_title":"迈向AI社会心理学：语言模型智能体再现类人的最小群体偏差","primary_category":"physics.soc-ph","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","社会心理学","群体偏差"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00009","has_summary":true},{"id":"2609.00345","title":"Do LLMs Know Your Neighborhood? Auditing LLM Priors for Neighborhood-Level Mobility Prediction and Structural Alignment","zh_title":"LLM了解你的社区吗？审计LLM先验用于社区级移动性预测与结构对齐","primary_category":"cs.LG","date":"2026-09-02","score":9,"bucket":"selected","tags":["LLM仿真","人类移动性","偏差审计"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00345","has_summary":true},{"id":"2609.00608","title":"Investigating Assistant Bias in LLM User Simulators Using a Role Vector","zh_title":"使用角色向量研究LLM用户模拟器中的助手偏差","primary_category":"cs.CL","date":"2026-09-02","score":8,"bucket":"selected","tags":["LLM用户模拟器","助手偏差","仿真有效性"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.00608","has_summary":true},{"id":"2609.00565","title":"Aligned but Flattened: Analyzing the Trade-off between Cultural Alignment and Diversity in LLMs","zh_title":"对齐但扁平化：分析LLMs中文化对齐与多样性之间的权衡","primary_category":"cs.SI","date":"2026-09-02","score":8,"bucket":"selected","tags":["文化仿真","算法保真度","价值观调查"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00565","has_summary":true},{"id":"2609.01519","title":"When Guardrails Look Effective: Construct Validity Failures in LLM Agent Commerce Evaluation","zh_title":"当护栏看似有效：LLM智能体商业评估中的构念效度失效","primary_category":"cs.AI","date":"2026-09-02","score":8,"bucket":"selected","tags":["LLM仿真","构念效度","市场模拟"],"rubric_hits":["A2","A4","B4"],"abs_url":"https://arxiv.org/abs/2609.01519","has_summary":true},{"id":"2609.00310","title":"Emotional Labor Strategy Preferences in LLM Personas","zh_title":"LLM人格中的情绪劳动策略偏好","primary_category":"cs.CL","date":"2026-09-02","score":7,"bucket":"pending","tags":["LLM仿真","情绪劳动","人格测量"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2609.00310","has_summary":true},{"id":"2609.00982","title":"Disclosure-Gated User Simulation for Companion-Agent Evaluation","zh_title":"面向陪伴智能体评估的披露门控用户仿真","primary_category":"cs.CL","date":"2026-09-02","score":7,"bucket":"pending","tags":["用户仿真","评估方法","偏差审计"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00982","has_summary":true},{"id":"2609.00250","title":"CompanionSim: Synthetic Data for Evaluating Anthropomorphism in Human-AI Relationships","zh_title":"CompanionSim：用于评估人机关系中拟人化的合成数据","primary_category":"cs.CY","date":"2026-09-02","score":7,"bucket":"pending","tags":["LLM仿真","人机交互","合成数据"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.00250","has_summary":true},{"id":"2609.01432","title":"Citing Less Critically: LLMs Reshape the Rhetoric and Reach of Scientific Citation","zh_title":"引用更少批判性：大语言模型重塑科学引用的修辞与影响范围","primary_category":"cs.DL","date":"2026-09-02","score":7,"bucket":"pending","tags":["LLM仿真","引用行为","科学计量"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2609.01432","has_summary":true},{"id":"2609.00248","title":"Authority Bias in Conversational Search Engines for Academic Paper Recommendation","zh_title":"学术论文推荐对话搜索引擎中的权威偏差","primary_category":"cs.AI","date":"2026-09-02","score":7,"bucket":"pending","tags":["LLM偏差","行为评估","推荐系统"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2609.00248","has_summary":true},{"id":"2608.18294","title":"Debiased Inference for AI-Generated Data without Gold-Standard Labels: Identification via Multiple Imperfect Measurements","zh_title":"无金标准标签下AI生成数据的去偏推断：基于多重不完美测量的识别","primary_category":"stat.ME","date":"2026-09-02","score":6,"bucket":"other","tags":["测量误差","统计推断","LLM标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.18294","has_summary":true},{"id":"2609.00940","title":"A Dataset for Modeling Iterative Problem-Solving","zh_title":"用于建模迭代问题求解的数据集","primary_category":"cs.CL","date":"2026-09-02","score":6,"bucket":"other","tags":["LLM预测人类行为","迭代问题求解","教育数据挖掘"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.00940","has_summary":false},{"id":"2609.01073","title":"Post-hoc Alignment of LLM-judges to Human Judgment Distribution","zh_title":"LLM评判者与人类判断分布的事后对齐","primary_category":"cs.CL","date":"2026-09-02","score":6,"bucket":"other","tags":["LLM评判","人类判断分布","事后对齐"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.01073","has_summary":false},{"id":"2609.00211","title":"AI Should Not Only Be Helpful. It Should Be Contingent. Artificial Intimacy, Sycophancy, and the Future of Social Learning","zh_title":"AI不应只是有用，而应具有条件性：人工亲密、谄媚与社会学习的未来","primary_category":"cs.AI","date":"2026-09-02","score":6,"bucket":"other","tags":["AI反馈","社会学习","人机交互"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00211","has_summary":false},{"id":"2609.00352","title":"How Does LGBTQIA+ Identity Affect LLM Behavior? Implications for Requirements Engineering of Mental Health AI Systems","zh_title":"LGBTQIA+身份如何影响LLM行为：对心理健康AI系统需求工程的启示","primary_category":"cs.CY","date":"2026-09-02","score":6,"bucket":"other","tags":["LLM公平性","身份披露","心理健康AI"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00352","has_summary":false},{"id":"2609.00373","title":"Corporate Loyalty: Some AI Systems Differentially Downplay their Creators' Controversies","zh_title":"企业忠诚度：一些AI系统对自身创造者的争议进行差异化淡化","primary_category":"cs.CY","date":"2026-09-02","score":6,"bucket":"other","tags":["模型偏差","企业忠诚","态度测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00373","has_summary":false},{"id":"2608.17809","title":"Whether LLMs Can Navigate Beliefs and Facts Depends on How You Phrase It","zh_title":"LLM能否驾驭信念与事实取决于提问方式","primary_category":"cs.CL","date":"2026-09-02","score":5,"bucket":"other","tags":["LLM评估","信念追踪","事实核查"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.17809","has_summary":false},{"id":"2608.18300","title":"The Lifecycle of LLM-as-a-Judge for Large-Scale Recommendation Explanations","zh_title":"大规模推荐解释中LLM评判者的生命周期","primary_category":"cs.AI","date":"2026-09-02","score":5,"bucket":"other","tags":["LLM-as-a-Judge","推荐系统","人类评估替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.18300","has_summary":false},{"id":"2609.00014","title":"Behaviorally Grounded User Profiles from the Wild for Personalized Alignment and Multi-Perspective Reasoning","zh_title":"基于真实行为数据的用户画像用于个性化对齐与多视角推理","primary_category":"cs.CL","date":"2026-09-02","score":5,"bucket":"other","tags":["用户画像","个性化对齐","多视角推理"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00014","has_summary":false},{"id":"2609.00491","title":"MemeBridge: A Dataset for Benchmarking and Mitigating the Bidirectional Cultural Gap in Meme Interpretation","zh_title":"MemeBridge：用于基准测试和缓解模因解读中双向文化差距的数据集","primary_category":"cs.CL","date":"2026-09-02","score":5,"bucket":"other","tags":["跨文化理解","数据集","LLM评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.00491","has_summary":false},{"id":"2609.00494","title":"Human-Anchored Factuality Evaluation with Strategic Annotation","zh_title":"基于策略标注的人类锚定事实性评估","primary_category":"cs.CL","date":"2026-09-02","score":5,"bucket":"other","tags":["事实性评估","人类标注","主动学习"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2609.00494","has_summary":false},{"id":"2609.00999","title":"Right Frame, Wrong Rule: Cultural Cues Expose the Financial Knowledge Gap They Were Meant to Close","zh_title":"正确的框架，错误的规则：文化线索暴露了它们本应缩小的金融知识差距","primary_category":"cs.CL","date":"2026-09-02","score":5,"bucket":"other","tags":["LLM评估","文化线索","金融知识"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00999","has_summary":false},{"id":"2609.00576","title":"Consistency Without Alignment: Item-Sensitive Language Models Indistinguishable From Random","zh_title":"无对齐的一致性：项目敏感语言模型与随机选择无异","primary_category":"cs.AI","date":"2026-09-02","score":5,"bucket":"other","tags":["LLM评估","项目敏感性","信号任务"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00576","has_summary":false},{"id":"2609.00304","title":"The Assistant's Ideal Self","zh_title":"助手的理想自我","primary_category":"cs.AI","date":"2026-09-02","score":5,"bucket":"other","tags":["LLM自我概念","价值观测量","模型对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2609.00304","has_summary":false},{"id":"2609.01167","title":"Classic AI Scaffolding for LLM Social Agents","zh_title":"面向LLM社会智能体的经典AI脚手架","primary_category":"cs.MA","date":"2026-09-02","score":5,"bucket":"other","tags":["社会模拟","多智能体","LLM架构"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2609.01167","has_summary":false},{"id":"2609.00453","title":"mimeo: Compiling Public Expert Corpora into Agent Skills and Testing What Transfers","zh_title":"mimeo：将公开专家语料编译为智能体技能并测试可迁移内容","primary_category":"cs.AI","date":"2026-09-02","score":3,"bucket":"other","tags":["专家人格","知识获取","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00453","has_summary":false},{"id":"2608.29215","title":"Attribute-Based Activation Steering of LLMs for Group-Specific Explanation Generation","zh_title":"基于属性的LLM激活引导用于群体特定解释生成","primary_category":"cs.CL","date":"2026-09-02","score":2,"bucket":"other","tags":["可解释性","激活引导","个性化生成"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.29215","has_summary":false},{"id":"2609.00063","title":"Medical Causal Hypothesis Verification with Large Language Models","zh_title":"用大语言模型验证医学因果假设","primary_category":"cs.CL","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM评测","因果推理","医疗信息检索"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00063","has_summary":false},{"id":"2609.00747","title":"Can Large Language Models Forecast What Researchers Study Next?","zh_title":"大语言模型能预测研究者下一步研究什么吗？","primary_category":"cs.CL","date":"2026-09-02","score":2,"bucket":"other","tags":["研究趋势预测","基准评测","LLM评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00747","has_summary":false},{"id":"2609.01548","title":"SDARE-Bench: Evaluating Large Language Models on Conversational Stigma Detection and Response in Dyadic and Group Dialogue","zh_title":"SDARE-Bench：评估大语言模型在二元与群体对话中的污名检测与回应","primary_category":"cs.CL","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM安全","对话评测","污名检测"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.01548","has_summary":false},{"id":"2609.00192","title":"LLM-Driven Autonomous Vehicles Inherit Human Driver Biases in Pedestrian Yielding: Results and Implications From A New Benchmark","zh_title":"LLM驱动的自动驾驶汽车在行人让行中继承人类驾驶员偏见：新基准的结果与启示","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["自动驾驶","偏见检测","视觉语言模型"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.00192","has_summary":false},{"id":"2609.00319","title":"Sources of Truth: A Multi-Platform, Multilingual Audit of Citations in AI Mental Health Information Queries","zh_title":"真相之源：AI心理健康信息查询中引文的多平台多语言审计","primary_category":"cs.CY","date":"2026-09-02","score":2,"bucket":"other","tags":["AI审计","健康信息","引文分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00319","has_summary":false},{"id":"2609.00921","title":"VIBE-Bench: Evaluating Personalized Large Language Models When Profiles Don't Mean Preferences","zh_title":"VIBE-Bench：当用户画像不等于偏好时评估个性化大语言模型","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["个性化LLM","偏好推理","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00921","has_summary":false},{"id":"2609.00334","title":"Human-AI Co-Interpretation for Responsible AI: A Hermeneutic Perspective","zh_title":"负责任AI的人机共同解释：诠释学视角","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["人机交互","诠释学","负责任AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00334","has_summary":false},{"id":"2609.00441","title":"Conversation Coach: A Voice-enabled AI System that Helps Practice Difficult Workplace Conversations","zh_title":"对话教练：一个语音AI系统，帮助练习困难的职场对话","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["语音对话系统","职场培训","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00441","has_summary":false},{"id":"2609.00652","title":"Self-Reports Are Not Verification: Environment-Grounded Auditing of LLM Operators in Evolutionary Search","zh_title":"自我报告并非验证：进化搜索中LLM操作者的环境接地审计","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM自我报告","进化搜索","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.00652","has_summary":false},{"id":"2609.00731","title":"Agentic Empirical Asset Pricing: Methodological Foundations","zh_title":"智能体实证资产定价：方法论基础","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM智能体","资产定价","因子发现"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.00731","has_summary":false},{"id":"2609.00904","title":"In-Context Neurofeedback: Can LLMs Control Their Internal Representations through Privileged Access?","zh_title":"上下文神经反馈：LLM能否通过特权访问控制其内部表征？","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM内部表征","神经反馈","元认知评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00904","has_summary":false},{"id":"2609.01337","title":"LEAP: Likelihood Elicitation and Aggregation for LLM-based Probabilistic Forecasting","zh_title":"LEAP：基于LLM的概率预测的似然启发与聚合","primary_category":"cs.AI","date":"2026-09-02","score":2,"bucket":"other","tags":["LLM预测","概率聚合","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01337","has_summary":false},{"id":"2609.00946","title":"Embedded Conditional Independence Tests for Large Language Model Generated Text with an Application to German Parliament Speeches","zh_title":"大语言模型生成文本的嵌入式条件独立性检验及其在德国议会演讲中的应用","primary_category":"stat.ML","date":"2026-09-02","score":2,"bucket":"other","tags":["条件独立性检验","LLM文本分析","统计方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00946","has_summary":false},{"id":"2609.01194","title":"Births are difficult to predict even with rich survey and full-population register data","zh_title":"即使有丰富的调查和全人口登记数据，生育也难以预测","primary_category":"cs.LG","date":"2026-09-02","score":2,"bucket":"other","tags":["预测建模","生育行为","数据挑战"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.01194","has_summary":false},{"id":"2608.19208","title":"When Irrelevant Text Matters: Affine Margin Shifts in Multimodal Large Language Models","zh_title":"当无关文本起作用：多模态大语言模型中的仿射边际偏移","primary_category":"cs.CL","date":"2026-09-02","score":0,"bucket":"other","tags":["多模态大语言模型","鲁棒性","决策边际"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.19208","has_summary":false},{"id":"2608.30303","title":"Lazy Grounding: Attacking Search Agents with Factual Evidence","zh_title":"惰性接地：用事实证据攻击搜索智能体","primary_category":"cs.CL","date":"2026-09-02","score":0,"bucket":"other","tags":["搜索智能体","对抗攻击","事实接地"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30303","has_summary":false},{"id":"2609.00191","title":"Assessing Suicide Risk in Arabic Crisis Helpline Calls: A Comparison of Arabic and English Large Language Models","zh_title":"评估阿拉伯语危机热线电话中的自杀风险：阿拉伯语与英语大语言模型的比较","primary_category":"cs.CL","date":"2026-09-02","score":0,"bucket":"other","tags":["自杀风险检测","NLP分类","危机热线"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00191","has_summary":false},{"id":"2609.01491","title":"GlossoGen: Emergent Language in Complex Multi-Agent LLM Interactions","zh_title":"GlossoGen：复杂多智能体LLM交互中的涌现语言","primary_category":"cs.CL","date":"2026-09-02","score":0,"bucket":"other","tags":["多智能体系统","语言演化","LLM交互"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01491","has_summary":false},{"id":"2609.01056","title":"WorldBench: Culturally Grounded Benchmark for Multilingual Agents","zh_title":"WorldBench：面向多语言智能体的文化基准","primary_category":"cs.AI","date":"2026-09-02","score":0,"bucket":"other","tags":["多智能体基准","任务执行","多语言"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2609.01056","has_summary":false},{"id":"2609.00100","title":"Different representation learning objectives recover distinct latent structures from the same psychometric data","zh_title":"不同表征学习目标从相同心理测量数据中恢复出不同的潜在结构","primary_category":"cs.AI","date":"2026-09-02","score":0,"bucket":"other","tags":["表征学习","心理测量","对比学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00100","has_summary":false},{"id":"2609.00180","title":"Asymmetries in Spontaneous and Instructed Deception","zh_title":"自发与指示欺骗中的不对称性","primary_category":"cs.AI","date":"2026-09-02","score":0,"bucket":"other","tags":["LLM欺骗","模型机制","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00180","has_summary":false},{"id":"2609.00584","title":"Socrates went Nuclear: Comparing Interaction Strategies for AI systems in a Learning Context using Brain Sensing","zh_title":"苏格拉底走向核能：在学习情境下使用脑传感比较AI系统的交互策略","primary_category":"cs.AI","date":"2026-09-02","score":0,"bucket":"other","tags":["AI交互","学习效果","脑机接口"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2609.00584","has_summary":false},{"id":"2609.00987","title":"On the Human and Computer Alignment of Attribute-Based Music Matches","zh_title":"基于属性的音乐匹配的人机对齐研究","primary_category":"cs.SD","date":"2026-09-02","score":0,"bucket":"other","tags":["音乐相似度","感知实验","生成式AI伦理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2609.00987","has_summary":false},{"id":"2609.00414","title":"LPG Subsidy Reform, Energy Compensation, and Social Risk in Bolivia: A Machine-Learning Agent-Based Microsimulation","zh_title":"玻利维亚液化石油气补贴改革、能源补偿与社会风险：基于机器学习的智能体微观模拟","primary_category":"econ.EM","date":"2026-09-02","score":0,"bucket":"other","tags":["智能体模拟","补贴改革","机器学习"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2609.00414","has_summary":false},{"id":"2606.30085","title":"Tastes without distinction: silicon samples and the synthetic construction of tastes","zh_title":"无差别的品味：硅样本与品味的合成建构","primary_category":"cs.CL","date":"2026-09-01","score":10,"bucket":"selected","tags":["硅采样","文化品味仿真","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.30085","has_summary":true},{"id":"2608.03044","title":"Emulate or Estimate? The Divergent Strengths of Base and Post-Trained Language Models for Opinion Simulation","zh_title":"仿真还是估计？基础与后训练语言模型在意见模拟中的不同优势","primary_category":"cs.CL","date":"2026-09-01","score":10,"bucket":"selected","tags":["LLM仿真","意见模拟","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.03044","has_summary":true},{"id":"2608.29455","title":"Item-Mean Surrogates: Why Richer Persona Data Fail to Improve LLMs as Human Surrogates","zh_title":"项目均值替代：为何更丰富的人物数据未能提升LLM作为人类替代品的表现","primary_category":"cs.CL","date":"2026-09-01","score":10,"bucket":"selected","tags":["LLM仿真","人类替代","算法保真度"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.29455","has_summary":true},{"id":"2608.30033","title":"\"Act Like a 5th Grader\" is Not Enough: Bounding Knowledge in LLM-Based User Simulators","zh_title":"“像五年级学生一样行动”还不够：在基于LLM的用户模拟器中界定知识","primary_category":"cs.CL","date":"2026-09-01","score":10,"bucket":"selected","tags":["LLM仿真","认知边界","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.30033","has_summary":true},{"id":"2608.28615","title":"Distributional Validity and Calibration of a Korean Synthetic Persona Panel for Digital and AI Service Use: A Secondary-Data Validation Against the Korea Media Panel Survey","zh_title":"韩国合成人面板在数字与AI服务使用上的分布效度与校准：基于韩国媒体面板调查的二手数据验证","primary_category":"cs.CY","date":"2026-09-01","score":10,"bucket":"selected","tags":["LLM仿真","合成人面板","外部效度"],"rubric_hits":["A1","A2","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.28615","has_summary":true},{"id":"2608.30522","title":"Tariff Threats, Macroeconomic Expectations, and Policy Communication Strategies: Experiments Based on a Multi-Agent System","zh_title":"关税威胁、宏观经济预期与政策沟通策略：基于多智能体系统的实验","primary_category":"econ.GN","date":"2026-09-01","score":10,"bucket":"selected","tags":["LLM仿真","宏观经济预期","政策沟通"],"rubric_hits":["A1","A3","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.30522","has_summary":true},{"id":"2608.26849","title":"LiveSim: Simulating Environment-Shaped Users in Multi-Agent Live-Stream Ecosystems","zh_title":"LiveSim：在多智能体直播生态系统中模拟受环境塑造的用户","primary_category":"cs.AI","date":"2026-09-01","score":9,"bucket":"selected","tags":["LLM用户仿真","直播生态","行为保真度"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.26849","has_summary":true},{"id":"2608.29803","title":"Do LLMs Change Their Minds Like Humans? Diagnosing Human--LLM Divergence in Single-Turn Persuasion Judgments","zh_title":"LLM会像人类一样改变想法吗？诊断单轮说服判断中的人机分歧","primary_category":"cs.CY","date":"2026-09-01","score":9,"bucket":"selected","tags":["LLM仿真","信念更新","人机对比"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.29803","has_summary":true},{"id":"2608.28668","title":"Reference-Distribution Dependence in LLM-Based Synthetic Persona Data: Diagnosis and Post Hoc Adjustment of Demographic Distributions","zh_title":"基于LLM的合成人数据中的参考分布依赖：人口统计分布的诊断与事后调整","primary_category":"cs.CY","date":"2026-09-01","score":9,"bucket":"selected","tags":["LLM仿真","人口统计偏差","事后加权调整"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.28668","has_summary":true},{"id":"2608.29266","title":"Measurement Validity in LLM Cultural Alignment","zh_title":"大语言模型文化对齐中的测量效度","primary_category":"physics.soc-ph","date":"2026-09-01","score":9,"bucket":"selected","tags":["LLM仿真","文化价值观","测量信度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.29266","has_summary":true},{"id":"2608.29535","title":"Integrating adaptive human behavior into epidemic models with large language models","zh_title":"用大语言模型将自适应人类行为整合进流行病模型","primary_category":"physics.soc-ph","date":"2026-09-01","score":9,"bucket":"selected","tags":["LLM仿真","流行病建模","政策评估"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.29535","has_summary":true},{"id":"2602.16061","title":"AI-Generated Measurements for Identification and Inference with Missing Data: A Weak Shadow Variable Approach","zh_title":"AI生成的测量用于缺失数据下的识别与推断：弱影子变量方法","primary_category":"stat.ML","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM生成测量","缺失数据","部分识别"],"rubric_hits":["A5","B1","B3"],"abs_url":"https://arxiv.org/abs/2602.16061","has_summary":true},{"id":"2608.01017","title":"Why LLMs Give In: Conversational Factors and Reasoning Behind Medical Sycophancy","zh_title":"为何大语言模型会屈服：医疗谄媚背后的对话因素与推理","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM可靠性","医疗问答","谄媚行为"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.01017","has_summary":true},{"id":"2608.25952","title":"Spatial-Knowledge-Graph-Grounded LLM Agents for Neighborhood Livability Evaluation","zh_title":"基于空间知识图谱的LLM智能体用于邻里宜居性评估","primary_category":"cs.CY","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM仿真","城市研究","智能体建模"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.25952","has_summary":true},{"id":"2608.28626","title":"Do large language models scrutinise what they review? A multimodal audit of scoring calibration, error detection, and author-identity effects","zh_title":"大语言模型会仔细审查它们所评审的内容吗？对评分校准、错误检测和作者身份效应的多模态审计","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM评审","可靠性评估","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.28626","has_summary":true},{"id":"2608.29446","title":"Whose Assessment of Distress? Community Perspectives and LLM Alignment on Well-Being Posts","zh_title":"谁的痛苦评估？社区视角与LLM在健康帖上的对齐","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","偏差评估"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.29446","has_summary":true},{"id":"2608.29453","title":"AI Can Be Easily Persuaded in Clinical Decision Making","zh_title":"AI在临床决策中容易被说服","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM决策","说服影响","临床AI"],"rubric_hits":["A1","B4"],"abs_url":"https://arxiv.org/abs/2608.29453","has_summary":true},{"id":"2608.29571","title":"Which one is banana man? Evaluating vision-language models in multi-turn pragmatic interpretation","zh_title":"谁是香蕉人？评估视觉-语言模型在多轮语用解释中的表现","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["语用推理","人类对照","视觉-语言模型"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.29571","has_summary":true},{"id":"2608.29995","title":"Generating Clinical Vignettes that Preserve Cognitive Formulations","zh_title":"生成保留认知公式的临床案例","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM生成","临床案例","认知模型"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.29995","has_summary":true},{"id":"2608.30110","title":"Can LLMs Take the Pulse of the Economy? A Real-Time Evaluation of LLM Nowcasts on Macroeconomic Indicators","zh_title":"LLM能否把握经济脉搏？对宏观经济指标实时预测的评估","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM仿真","宏观经济预测","实时评估"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.30110","has_summary":true},{"id":"2608.30873","title":"Personas Differ from Native-Language Generation: Language Pathways Shape LLM Interpersonal Advice","zh_title":"人设与母语生成不同：语言路径塑造LLM的人际建议","primary_category":"cs.CL","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM仿真","跨语言行为","方法偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.30873","has_summary":true},{"id":"2608.31059","title":"When Can We Work in Embedding Space? What Text Embeddings Preserve","zh_title":"何时可以在嵌入空间中工作？文本嵌入保留了什么","primary_category":"econ.EM","date":"2026-09-01","score":7,"bucket":"pending","tags":["文本嵌入","LLM生成文本","经济数据分析"],"rubric_hits":["A5","B1"],"abs_url":"https://arxiv.org/abs/2608.31059","has_summary":true},{"id":"2608.30210","title":"Frontier vision-language models have overtaken young adults at detecting AI-generated portraits -- but not their calibration","zh_title":"前沿视觉语言模型在检测AI生成人像上已超越年轻人——但校准能力尚未超越","primary_category":"cs.HC","date":"2026-09-01","score":7,"bucket":"pending","tags":["视觉语言模型","人类对照","校准偏差"],"rubric_hits":["A2","B1"],"abs_url":"https://arxiv.org/abs/2608.30210","has_summary":true},{"id":"2608.30311","title":"One AI Signal, Many Human Judgments: A Bayesian Cascade Analysis of AI-based Credibility Indicators in Online Information Spread","zh_title":"一个AI信号，多种人类判断：在线信息传播中基于AI的可信度指标的贝叶斯级联分析","primary_category":"cs.HC","date":"2026-09-01","score":7,"bucket":"pending","tags":["人类-AI交互","社会学习","信息传播"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.30311","has_summary":true},{"id":"2608.28597","title":"The Race between Agentic AI Capabilities and Data Quality Control in Online Surveys","zh_title":"在线调查中代理式AI能力与数据质量控制之间的竞赛","primary_category":"cs.AI","date":"2026-09-01","score":7,"bucket":"pending","tags":["LLM代理","调查数据质量","注意力检查"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.28597","has_summary":true},{"id":"2608.07438","title":"PsychoAgent: An Affect-Sensitive Cognitive Architecture for Conflict-Aware Memory in LLM Agents","zh_title":"PsychoAgent：一种面向LLM智能体的情感敏感认知架构，用于冲突感知记忆","primary_category":"cs.AI","date":"2026-09-01","score":6,"bucket":"other","tags":["LLM智能体","认知架构","情感记忆"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.07438","has_summary":false},{"id":"2608.20983","title":"Beyond Truth Discovery: A Two-Stage Framework to Assess the Severity of False Claim during Disasters","zh_title":"超越真相发现：评估灾害期间虚假声明严重性的两阶段框架","primary_category":"cs.SI","date":"2026-09-01","score":6,"bucket":"other","tags":["虚假信息","人机对齐","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.20983","has_summary":false},{"id":"2608.29209","title":"Toward Cultural Alignment: Human-Centered Evaluation of Multimodal AI Stories Across Five African Communities","zh_title":"迈向文化对齐：对五个非洲社区的多模态AI故事进行以人为中心的评估","primary_category":"cs.CL","date":"2026-09-01","score":6,"bucket":"other","tags":["文化对齐","多模态评估","LLM评判"],"rubric_hits":["D1","B1"],"abs_url":"https://arxiv.org/abs/2608.29209","has_summary":false},{"id":"2608.29517","title":"LLM Judges as Raters: A Pre-Registered Audit of Severity, Halo, Reliability, and Version Instability in LLM Essay Scoring on Public Corpora","zh_title":"LLM法官作为评分者：对公共语料库中LLM作文评分的严重性、晕轮效应、可靠性和版本不稳定性的预注册审计","primary_category":"cs.CL","date":"2026-09-01","score":6,"bucket":"other","tags":["LLM评分","评分者效应","教育测量"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.29517","has_summary":false},{"id":"2608.30373","title":"Beyond Consensus: Downward Bias and Role Asymmetry in Multi-Agent LLM Judges for Subjective Evaluation","zh_title":"超越共识：多智能体LLM评判在主观评估中的向下偏差与角色不对称性","primary_category":"cs.CL","date":"2026-09-01","score":6,"bucket":"other","tags":["LLM评估","多智能体","人类对齐"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.30373","has_summary":false},{"id":"2608.29198","title":"How Identity and Opinion Shape Political Sycophancy in LLMs","zh_title":"身份与观点如何塑造大语言模型的政治谄媚行为","primary_category":"cs.AI","date":"2026-09-01","score":6,"bucket":"other","tags":["LLM政治立场","谄媚行为","模型偏差"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.29198","has_summary":false},{"id":"2608.28989","title":"Using LLMs to Mimic the Conversational Dynamics of Reddit Communities","zh_title":"使用大语言模型模仿Reddit社区的对话动态","primary_category":"cs.HC","date":"2026-09-01","score":6,"bucket":"other","tags":["LLM仿真","社交媒体","风格模仿"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.28989","has_summary":false},{"id":"2607.26178","title":"DuplexGen: Adaptive Synthesis of Human-AI Turn-Taking Dialogues","zh_title":"DuplexGen：自适应合成人机轮流对话","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["对话生成","人类偏好校准","轮流行为"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.26178","has_summary":false},{"id":"2608.24080","title":"When Less Is More: An Empirical Study of Minimal Responses in Counseling Dialogues and the Behavior of LLMs","zh_title":"少即是多：咨询对话中最小回应及LLM行为的实证研究","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM评估","心理咨询对话","最小回应"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24080","has_summary":false},{"id":"2608.26123","title":"Which India Survives Translation? Narrative Homogenisation Across Indian Oral Traditions in LLMs","zh_title":"哪个印度在翻译中幸存？LLM中印度口头传统的叙事同质化","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["文化表征","叙事同质化","LLM偏差"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.26123","has_summary":false},{"id":"2608.28649","title":"Can Large Language Models Identify Meaningful Touchpoints in Conversion Attribution?","zh_title":"大语言模型能否识别转化归因中有意义的触点？","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["转化归因","LLM标注","语义关联"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.28649","has_summary":false},{"id":"2608.29591","title":"How You Ask Shapes What You Get: A Theory-Seeded Measurement of Articulation in Advice-Seeking LLM Conversations","zh_title":"提问方式塑造回答：基于理论种子测量建议寻求型LLM对话中的表达清晰度","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM行为分析","提示词风格","对话测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.29591","has_summary":false},{"id":"2608.29738","title":"Evaluating the Capabilities of LLMs for Persuasive Dialogue","zh_title":"评估大语言模型在说服性对话中的能力","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM辩论","论证理论","人机对比"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.29738","has_summary":false},{"id":"2608.30224","title":"The Differential Reasoning Router: Operationalizing Cost-Aware LLM Annotation in E-commerce","zh_title":"差分推理路由器：在电商中实现成本感知的LLM标注","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM标注","成本优化","人机协作"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.30224","has_summary":false},{"id":"2608.30485","title":"Two Centuries of Sexism in British Parliament: A Computational Analysis of Women's Representation in the Hansard Corpus","zh_title":"英国议会两个世纪的性别歧视：对汉萨德语料库中女性代表性的计算分析","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM标注","性别歧视","议会辩论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.30485","has_summary":false},{"id":"2608.30683","title":"WildSEEK: Evaluating Language Models for Information-Seeking","zh_title":"WildSEEK：评估语言模型的信息寻求行为","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM评估","信息寻求","风险与公平"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.30683","has_summary":false},{"id":"2608.30754","title":"CLIN: an Objective Framework for Evaluating Creativity in Short Persian Literary Text","zh_title":"CLIN：评估波斯语短文学文本创造力的客观框架","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM评估","创造力测量","低资源语言"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.30754","has_summary":false},{"id":"2608.30842","title":"Thesis Proposal: Toward a Human-Centered and Perspective-Aware Framework for Reproducible ML Evaluation and AI Alignment","zh_title":"面向可复现机器学习评估与AI对齐的以人为本、视角感知框架","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["AI评估","人类分歧","对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.30842","has_summary":false},{"id":"2608.30902","title":"Low-Resource Preference Adaptation of LLMs via Activation-Based Label Propagation","zh_title":"基于激活标签传播的低资源LLM偏好适配","primary_category":"cs.CL","date":"2026-09-01","score":5,"bucket":"other","tags":["偏好优化","低资源标注","激活探针"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.30902","has_summary":false},{"id":"2608.30023","title":"Demand-Side Measurement for Generative Engine Optimization: Constructing and Validating a Million-Persona, Intent-Annotated Buyer Corpus","zh_title":"生成引擎优化的需求侧测量：构建并验证百万级意图标注买家画像语料库","primary_category":"cs.IR","date":"2026-09-01","score":5,"bucket":"other","tags":["合成数据","买家画像","生成引擎优化"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.30023","has_summary":false},{"id":"2608.28979","title":"AREAs-Lab: An Interactive Environment for AI-driven Requirement Elicitation for AI Systems","zh_title":"AREAs-Lab：面向AI系统的AI驱动需求获取交互环境","primary_category":"cs.HC","date":"2026-09-01","score":5,"bucket":"other","tags":["需求获取","AI模拟用户","交互评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.28979","has_summary":false},{"id":"2608.29292","title":"Measuring the \"Interaction Gap\" in Drama Therapy with AI","zh_title":"测量戏剧治疗中与AI的“互动差距”","primary_category":"cs.HC","date":"2026-09-01","score":5,"bucket":"other","tags":["AI与人类对比","戏剧治疗","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.29292","has_summary":false},{"id":"2608.29306","title":"The relationship between professional and general ethics in generative AI","zh_title":"生成式AI中职业伦理与一般伦理的关系","primary_category":"cs.CY","date":"2026-09-01","score":5,"bucket":"other","tags":["AI伦理","职业伦理","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.29306","has_summary":false},{"id":"2608.28628","title":"CDEP Agent: Connecting Meteorologically Detected Temporal Compound Events to Real-World Documentary Evidence","zh_title":"CDEP Agent：将气象检测的复合事件与真实世界文献证据相连接","primary_category":"cs.AI","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM agent","气候事件","证据链接"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.28628","has_summary":false},{"id":"2608.28644","title":"Measuring Collective Semantic Change in Populations of Language Model Agents","zh_title":"测量语言模型智能体群体中的集体语义变化","primary_category":"physics.soc-ph","date":"2026-09-01","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","语义变化"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.28644","has_summary":false},{"id":"2608.28703","title":"Save 2050: A Planetary-Scale Collective Prediction System for the Singularity Crisis","zh_title":"拯救2050：应对奇点危机的行星级集体预测系统","primary_category":"physics.soc-ph","date":"2026-09-01","score":5,"bucket":"other","tags":["集体预测","社会模拟","未来研究"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.28703","has_summary":false},{"id":"2607.23648","title":"EmoTrace: An Emotion Trajectory-Centered Framework for Psychological Support Dialogue Generation","zh_title":"EmoTrace：以情绪轨迹为中心的心理支持对话生成框架","primary_category":"cs.CL","date":"2026-09-01","score":3,"bucket":"other","tags":["对话生成","心理咨询","情绪建模"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.23648","has_summary":false},{"id":"2608.29481","title":"SIC-Agents: Benchmarking and Building an Adaptive Simulator for Pediatric Serious Illness Communication Training","zh_title":"SIC-Agents：面向儿科重症沟通训练的自适应模拟器基准与构建","primary_category":"cs.CL","date":"2026-09-01","score":3,"bucket":"other","tags":["对话模拟","医学培训","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.29481","has_summary":false},{"id":"2608.30948","title":"Detecting AI Impostors: How Do Middle Schoolers Identify LLM Agents in a Live Collaborative Setting?","zh_title":"检测AI冒充者：中学生在实时协作环境中如何识别LLM智能体？","primary_category":"cs.CL","date":"2026-09-01","score":3,"bucket":"other","tags":["AI冒充检测","人机交互","教育技术"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.30948","has_summary":false},{"id":"2608.30694","title":"Inferring Value Criteria from Ordinal Preferences: An Iterative In-Context Learning Framework for Music Generation","zh_title":"从序数偏好推断价值标准：面向音乐生成的迭代式上下文学习框架","primary_category":"cs.HC","date":"2026-09-01","score":3,"bucket":"other","tags":["LLM音乐生成","偏好学习","模拟评分者"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30694","has_summary":false},{"id":"2607.28648","title":"Why It Hurts: Identifying the Drivers of Negative Thoughts in Emotional Support Conversations","zh_title":"为何痛苦：识别情感支持对话中负面思维驱动因素","primary_category":"cs.HC","date":"2026-09-01","score":2,"bucket":"other","tags":["情感支持","认知评估","多智能体框架"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28648","has_summary":false},{"id":"2607.29181","title":"SERUM: State Extraction and Refinement for User Modeling","zh_title":"SERUM：面向用户建模的状态提取与精炼","primary_category":"cs.LG","date":"2026-09-01","score":2,"bucket":"other","tags":["用户建模","视频理解","行为分析"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.29181","has_summary":false},{"id":"2608.28619","title":"From GenAI Virtual Patient Dialogue Logs to Teacher-Interpretable Process Evidence: A Learning Analytics Study in Higher Education","zh_title":"从生成式AI虚拟患者对话日志到教师可解释的过程证据：高等教育中的学习分析研究","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["医学教育","虚拟患者","学习分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28619","has_summary":false},{"id":"2608.28623","title":"Looking Again: Measuring Sycophancy in the Reasoning Chains of Multimodal Models Under Pressure","zh_title":"再看：压力下多模态模型推理链中的谄媚测量","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["多模态模型","谄媚行为","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.28623","has_summary":false},{"id":"2608.29492","title":"CoCoA: Context-Conditional Cultural Alignment for Large Language Models","zh_title":"CoCoA：大语言模型的上下文条件文化对齐","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["文化偏见","模型对齐","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29492","has_summary":false},{"id":"2608.29582","title":"SUP-MIMIC: A Multi-Task Clinical Diagnosis Benchmark for Evaluating LLMs' Robustness to Contradictory Evidence","zh_title":"SUP-MIMIC：评估大语言模型对矛盾证据鲁棒性的多任务临床诊断基准","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["LLM评测","临床诊断","推理鲁棒性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29582","has_summary":false},{"id":"2608.29610","title":"Beyond Surface Alignment: Grounding the Dynamics of Situational Understanding and Generative Control in LLMs","zh_title":"超越表面对齐：在LLM中奠定情境理解与生成控制的动力学基础","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["LLM对齐","情境理解","生成控制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29610","has_summary":false},{"id":"2608.29798","title":"R$^2$A: Learning Persona Policies Through Persona Representation Learning and Runtime Alignment","zh_title":"R²A：通过人格表征学习与运行时对齐学习人格策略","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["角色扮演","人格策略","对齐"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.29798","has_summary":false},{"id":"2608.30065","title":"Pak3H: Evaluating the Cost of Cultural Mismatch in LLM Alignment with a Human-Contextualized Urdu Benchmark","zh_title":"Pak3H：用人类情境化乌尔都语基准评估LLM对齐中的文化错配成本","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["LLM对齐","多语言基准","文化适应性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30065","has_summary":false},{"id":"2608.30086","title":"When Does a Classifier Help an LLM? Classifier-Guided Prompting and Hybrid Classifier-LLM Models for Credit-Default Prediction","zh_title":"分类器何时能帮助大语言模型？分类器引导提示与混合分类器-LLM模型用于信用违约预测","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["信用违约预测","分类器-LLM混合","提示工程"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30086","has_summary":false},{"id":"2608.30241","title":"PaperBanana-Interact: Scientific Diagram Refinement with Multi-Turn Human Feedback","zh_title":"PaperBanana-Interact：基于多轮人类反馈的科学图表精化","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["多智能体系统","图表生成","用户模拟器"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30241","has_summary":false},{"id":"2608.30297","title":"AIA$^{2}$: Attribute-Agnostic Imbalance Augmentation for Subgroup Robustness","zh_title":"AIA²：面向子群鲁棒性的属性无关不平衡增强","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["数据增强","子群鲁棒性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30297","has_summary":false},{"id":"2608.30716","title":"SocialReasonBench: A Video-QA Benchmark for Social Reasoning with Counterfactual Narrative Videos","zh_title":"SocialReasonBench：基于反事实叙事视频的社会推理视频问答基准","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["视频问答基准","社会推理","多模态模型评测"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.30716","has_summary":false},{"id":"2608.30924","title":"TRIPPULSE: Multi-Agent Travel Planning with Review-Grounded Reasoning","zh_title":"TRIPPULSE：基于评论推理的多智能体旅行规划","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["多智能体系统","旅行规划","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30924","has_summary":false},{"id":"2608.30980","title":"Evaluating and Improving LLM Self-Modeling","zh_title":"评估与改进大语言模型的自我建模能力","primary_category":"cs.CL","date":"2026-09-01","score":2,"bucket":"other","tags":["LLM自我建模","能力评测","合成数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30980","has_summary":false},{"id":"2608.28837","title":"Delegating Before Learning: Where Generative AI Sits in Students' Professional Communication","zh_title":"学习前委托：生成式AI在学生专业沟通中的位置","primary_category":"cs.HC","date":"2026-09-01","score":2,"bucket":"other","tags":["生成式AI使用","人机交互","教育技术"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28837","has_summary":false},{"id":"2608.28604","title":"The Brand War: A Gamified AI-Feedback System for Time-Limited EFL Writing","zh_title":"品牌之战：限时EFL写作的游戏化AI反馈系统","primary_category":"cs.CY","date":"2026-09-01","score":2,"bucket":"other","tags":["AI反馈","游戏化学习","EFL写作"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28604","has_summary":false},{"id":"2608.29174","title":"Sustained Heterogeneity: an emergent collective mechanism in LLM-driven traffic","zh_title":"持续异质性：LLM驱动交通中的涌现集体机制","primary_category":"physics.soc-ph","date":"2026-09-01","score":2,"bucket":"other","tags":["LLM控制","交通仿真","多智能体"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.29174","has_summary":false},{"id":"2608.30946","title":"Reproducible macroscopic dynamics in a closed-loop human-AI learning system","zh_title":"闭环人机学习系统中的可复现宏观动力学","primary_category":"cs.LG","date":"2026-09-01","score":2,"bucket":"other","tags":["人机交互","学习系统","宏观动力学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30946","has_summary":false},{"id":"2608.17054","title":"Why This and Not That? A Collaborative Reflection Approach for Understanding Thought Coverage in Decision Making Support Dialog","zh_title":"为何此而非彼？一种用于理解决策支持对话中思维覆盖的协作反思方法","primary_category":"cs.HC","date":"2026-09-01","score":0,"bucket":"other","tags":["对话代理","决策支持","用户研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.17054","has_summary":false},{"id":"2608.18795","title":"Decomposing Wrong-Consensus Agreement in LLM Self-Consistency","zh_title":"分解LLM自一致性中的错误共识一致性","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["LLM可靠性","自一致性","错误共识"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.18795","has_summary":false},{"id":"2608.28633","title":"PAUSE: Editable Strategy Artifacts for Long-Form Cultural Story Adaptation","zh_title":"PAUSE：用于长篇文化故事改编的可编辑策略工件","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["文化改编","人机交互","故事生成"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28633","has_summary":false},{"id":"2608.29109","title":"Recognition-Refusal Misalignment in LLMs: Why Models Answer Structurally Unanswerable Questions","zh_title":"大语言模型中的识别-拒绝错位：为何模型回答结构上不可回答的问题","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["模型行为分析","拒绝机制","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29109","has_summary":false},{"id":"2608.29257","title":"Large Language Models Systematically Favor Popular Options: Evidence and Mitigation Across MCQs","zh_title":"大语言模型系统性偏好流行选项：来自多选题的证据与缓解方法","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["LLM评测","多选题偏差","推理时校正"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29257","has_summary":false},{"id":"2608.30256","title":"Beyond Surface Forms: Symbolic Edits as a Test for Logical Reasoning with LLMs","zh_title":"超越表面形式：符号编辑作为大语言模型逻辑推理的测试","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["逻辑推理","模型评测","符号编辑"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30256","has_summary":false},{"id":"2608.30270","title":"Read the Room, Read the Image: Understanding Indirect Speech Acts in Multimodal Visual Contexts","zh_title":"察言观色：理解多模态视觉语境中的间接言语行为","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["多模态基准","语用推理","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.30270","has_summary":false},{"id":"2608.30391","title":"Using Grounded Theory for Agent Behavior Analysis at Scale","zh_title":"使用扎根理论进行大规模智能体行为分析","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["智能体行为分析","扎根理论","轨迹挖掘"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30391","has_summary":false},{"id":"2608.31035","title":"When Does Predictor-Based RL Align with Human Perception? A Study of Subjective Rewards in Codec-Based Speech Language Models","zh_title":"基于预测器的强化学习何时与人类感知对齐？基于编解码语音语言模型的主观奖励研究","primary_category":"cs.CL","date":"2026-09-01","score":0,"bucket":"other","tags":["语音合成","强化学习","人类感知对齐"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.31035","has_summary":false},{"id":"2608.29118","title":"Emergent Misalignment Is Not Magical","zh_title":"涌现性失准并非魔法","primary_category":"cs.AI","date":"2026-09-01","score":0,"bucket":"other","tags":["AI安全","模型泛化","表征分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29118","has_summary":false},{"id":"2608.30052","title":"The Language of the Question Selects the Market: Query Language and Exit IP as Separable Factors in Commercial Recommendations from a Generative Search Interface","zh_title":"问题语言选择市场：生成式搜索界面商业推荐中查询语言与出口IP作为可分离因素","primary_category":"cs.IR","date":"2026-09-01","score":0,"bucket":"other","tags":["生成式搜索","商业推荐","语言效应"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30052","has_summary":false},{"id":"2608.30971","title":"The Hermon Moment: AI Self-Transcendence and Its Human Narration","zh_title":"赫尔蒙时刻：AI 自我超越及其人类叙事","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["多智能体系统","AI社会秩序","哲学叙事"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30971","has_summary":false},{"id":"2608.31007","title":"Augmenting Interviewer Judgments of Patient Experience with Automatic Language Analysis","zh_title":"用自动语言分析增强访谈者对患者体验的判断","primary_category":"cs.HC","date":"2026-09-01","score":0,"bucket":"other","tags":["自动语言分析","患者体验预测","临床访谈"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.31007","has_summary":false},{"id":"2608.29950","title":"The Policy Deficit in AI x Social-Emotional Learning Research","zh_title":"AI与社会情感学习研究中的政策缺失","primary_category":"cs.HC","date":"2026-09-01","score":0,"bucket":"other","tags":["系统综述","教育政策","AI与SEL"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.29950","has_summary":false},{"id":"2608.30424","title":"Towards Cognitive Process-Aware Proactive Writing Support","zh_title":"迈向认知过程感知的主动式写作支持","primary_category":"cs.HC","date":"2026-09-01","score":0,"bucket":"other","tags":["写作支持","认知过程","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.30424","has_summary":false},{"id":"2608.28621","title":"Experts Disagree on How to Fight AI Disinformation, but Agree That Health and Politics Need Different Solutions","zh_title":"专家对如何应对AI虚假信息意见不一，但一致认为健康和政治领域需要不同解决方案","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["AI虚假信息","专家调查","政策建议"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.28621","has_summary":false},{"id":"2608.28617","title":"Can AI-Assisted Inquiry Enhance Students' Decision-Making Skills in Socio-Scientific Issues? A Three-Group Experimental Study on Climate Change","zh_title":"AI辅助探究能否提升学生社会性科学议题决策能力？一项关于气候变化的三组实验研究","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["AI辅助教学","科学教育","决策能力"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28617","has_summary":false},{"id":"2608.29055","title":"Why Organizational Rules Fail AI: O-I-B-A-R and the Externalization of Decision Boundaries","zh_title":"为何组织规则在AI中失效：O-I-B-A-R与决策边界的外化","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["组织AI","知识表示","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29055","has_summary":false},{"id":"2608.29751","title":"Large-Scale Qualitative Research with AI: Infrastructure, Management and Operation of the Socioscope Data Pipeline","zh_title":"AI驱动的大规模定性研究：Socioscope数据管道的基础设施、管理与运营","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["AI辅助定性研究","数据管道","食品系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29751","has_summary":false},{"id":"2608.29912","title":"Verification-Time Dependency on a Disappearing Evaluator","zh_title":"消失的评估者：验证时依赖性问题","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["AI治理","模型验证","可审计性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29912","has_summary":false},{"id":"2608.30084","title":"AMINA: The Inclusive and Accountable AI for Marginalized Immigrant Nonprofit Assistance","zh_title":"AMINA：面向边缘化移民非营利组织的包容且可问责的AI助手","primary_category":"cs.CY","date":"2026-09-01","score":0,"bucket":"other","tags":["AI助手","非营利组织","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.30084","has_summary":false},{"id":"2608.28631","title":"CrossAudit: A Git-Native, Cross-Vendor Audit Loop for Agentic Science","zh_title":"CrossAudit：面向智能体科学的 Git 原生跨供应商审计循环","primary_category":"cs.AI","date":"2026-09-01","score":0,"bucket":"other","tags":["多智能体系统","审计协议","AI 治理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.28631","has_summary":false},{"id":"2608.30938","title":"Evidence, Logic, and Compliance: Multi-Agent Structured Graph Reasoning with Expert Arbitration for Medical Referral","zh_title":"证据、逻辑与合规：基于专家仲裁的多智能体结构化图推理用于医疗转诊","primary_category":"cs.MA","date":"2026-09-01","score":0,"bucket":"other","tags":["多智能体系统","医疗决策","图推理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30938","has_summary":false},{"id":"2608.28606","title":"Cognitive Cells: A Compositional Framework for Populations of Small Language Models","zh_title":"认知细胞：小语言模型群体的组合框架","primary_category":"physics.soc-ph","date":"2026-09-01","score":0,"bucket":"other","tags":["多智能体系统","小语言模型","集体智能"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.28606","has_summary":false},{"id":"2608.29070","title":"Selective Disclosure of Hidden Directives in Reasoning Models: Behavioral Asymmetry and Steering","zh_title":"推理模型中隐藏指令的选择性披露：行为不对称性与操控","primary_category":"cs.LG","date":"2026-09-01","score":0,"bucket":"other","tags":["推理模型","指令遵循","AI安全"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29070","has_summary":false},{"id":"2608.29674","title":"Creation begins with understanding: LLMs as strategy designers for privacy-preserving tabular data synthesis","zh_title":"创造始于理解：LLM作为隐私保护表格数据合成的策略设计者","primary_category":"cs.LG","date":"2026-09-01","score":0,"bucket":"other","tags":["数据合成","隐私保护","LLM应用"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.29674","has_summary":false},{"id":"2608.30976","title":"A Human-in-the-Loop Autonomous Agent for Industry Time Series Forecasting","zh_title":"一种用于工业时间序列预测的人在回路自主智能体","primary_category":"cs.LG","date":"2026-09-01","score":0,"bucket":"other","tags":["时间序列预测","人在回路","自主智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.30976","has_summary":false},{"id":"2608.28620","title":"Preference Elicitation for Policy Optimization and Application to Aligning Heart Transplantation with Human Values","zh_title":"面向政策优化的偏好诱导及其在心脏移植与人类价值观对齐中的应用","primary_category":"cs.AI","date":"2026-09-01","score":0,"bucket":"other","tags":["偏好诱导","政策优化","器官分配"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.28620","has_summary":false},{"id":"2608.29097","title":"Optimally Selecting Representative Agents from a Metric Space","zh_title":"从度量空间中最优选择代表性智能体","primary_category":"cs.GT","date":"2026-09-01","score":0,"bucket":"other","tags":["公平聚类","算法博弈论","计算社会选择"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.29097","has_summary":false},{"id":"2608.28182","title":"Benchmarking large language model agent societies against human behavioural distributions","zh_title":"基于人类行为分布基准测试大语言模型智能体社会","primary_category":"physics.soc-ph","date":"2026-08-31","score":10,"bucket":"selected","tags":["LLM仿真","人类行为对照","算法保真度"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.28182","has_summary":true},{"id":"2608.26086","title":"TraceML: An Empirical Analysis of Human-Agent Planning in Machine Learning Development","zh_title":"TraceML：机器学习开发中人机规划的经验分析","primary_category":"cs.LG","date":"2026-08-31","score":7,"bucket":"pending","tags":["LLM agent","人类行为对照","过程分析"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.26086","has_summary":true},{"id":"2608.26152","title":"AI Models Can Predict and Collaboratively Modulate Human Memory Search","zh_title":"AI模型可以预测并协同调节人类记忆搜索","primary_category":"cs.CL","date":"2026-08-31","score":7,"bucket":"pending","tags":["LLM仿真","认知行为","人类数据对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.26152","has_summary":true},{"id":"2608.27465","title":"The Effect of Emotional Context on Large Language Models' Endorsement of Premature Decisions: Comparing Emotional Vulnerability Across Six Commercial Models","zh_title":"情绪语境对大语言模型认可过早决策的影响：六种商业模型情绪脆弱性比较","primary_category":"cs.CL","date":"2026-08-31","score":7,"bucket":"pending","tags":["LLM决策偏差","情绪影响","模型安全性"],"rubric_hits":["A1","B4"],"abs_url":"https://arxiv.org/abs/2608.27465","has_summary":true},{"id":"2608.28576","title":"Learning a Size-Weight Frontier for Synthetic-Augmented Inference","zh_title":"学习合成增强推断的规模-权重前沿","primary_category":"stat.ME","date":"2026-08-31","score":7,"bucket":"pending","tags":["合成数据","统计推断","LLM增强"],"rubric_hits":["A5","B1","B3"],"abs_url":"https://arxiv.org/abs/2608.28576","has_summary":true},{"id":"2608.28001","title":"FocusGen: Expanding Visual Design Exploration with a Simulated Focus Group of Persona Agents","zh_title":"FocusGen：用模拟焦点小组扩展视觉设计探索","primary_category":"cs.HC","date":"2026-08-31","score":7,"bucket":"pending","tags":["LLM仿真","人机交互","设计探索"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.28001","has_summary":true},{"id":"2608.27974","title":"QUORUM: QUality-Optimized Routing Using Multiple annotators","zh_title":"QUORUM：使用多标注者的质量优化路由","primary_category":"cs.CL","date":"2026-08-31","score":5,"bucket":"other","tags":["LLM标注","预算路由","数据标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.27974","has_summary":false},{"id":"2608.28144","title":"The Shape of Power: A Multilingual Framework for Social Power Reasoning in Dialogues","zh_title":"权力的形态：对话中社会权力推理的多语言框架","primary_category":"cs.AI","date":"2026-08-31","score":5,"bucket":"other","tags":["社会权力推理","多语言评估","LLM社会认知"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.28144","has_summary":false},{"id":"2608.28399","title":"RetailAgent: Structured Adverse Timing in Self-Conditioned Multimodal LLM Trading Agents","zh_title":"RetailAgent：自条件多模态LLM交易代理中的结构化逆向择时","primary_category":"cs.AI","date":"2026-08-31","score":5,"bucket":"other","tags":["LLM代理","金融决策","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.28399","has_summary":false},{"id":"2608.28378","title":"PersonaForge: Realistic Multi-Turn User Simulation for Agentic Systems","zh_title":"PersonaForge：面向智能体系统的真实多轮用户仿真","primary_category":"cs.CL","date":"2026-08-31","score":3,"bucket":"other","tags":["用户仿真","多智能体系统","智能体训练"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.28378","has_summary":false},{"id":"2608.28405","title":"CultureConverse: A Multilingual Multi-turn Simulation Harness for Culturally Grounded Assistance in East and Southeast Asia","zh_title":"CultureConverse：面向东亚与东南亚文化基础辅助的多语言多轮仿真与评估框架","primary_category":"cs.CL","date":"2026-08-31","score":3,"bucket":"other","tags":["文化对话评估","多语言数据集","LLM基准"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28405","has_summary":false},{"id":"2608.27843","title":"Synthetic Linguistic Agency: How an Embodied Mortal Agent Learns Linguistic Affordances through Consequential Social Experience","zh_title":"合成语言能动性：具身有死代理如何通过后果性社会经验学习语言可供性","primary_category":"cs.CL","date":"2026-08-31","score":2,"bucket":"other","tags":["具身智能体","语言学习","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27843","has_summary":false},{"id":"2608.27992","title":"GOD: Govern, Observe, and Direct - A Real-Time Control Room for Agent Societies","zh_title":"GOD：治理、观察与指导——面向智能体社会的实时控制室","primary_category":"cs.AI","date":"2026-08-31","score":2,"bucket":"other","tags":["多智能体系统","控制室","生成式智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27992","has_summary":false},{"id":"2608.27998","title":"Automated Analysis Framework for Multilingual Climate-Health Literature Based on Multi-Agent Large Language Model","zh_title":"基于多智能体大语言模型的多语言气候健康文献自动分析框架","primary_category":"cs.AI","date":"2026-08-31","score":2,"bucket":"other","tags":["多智能体系统","文献挖掘","气候健康"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27998","has_summary":false},{"id":"2608.28228","title":"Generative AI Alignment with Hinduism's Theological Plurality and Sacred Representation","zh_title":"生成式AI与印度教神学多元性及神圣表征的对齐","primary_category":"cs.AI","date":"2026-08-31","score":2,"bucket":"other","tags":["AI伦理","宗教互动","用户研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28228","has_summary":false},{"id":"2608.27927","title":"Antipatterns in AI-assisted Qualitative Data Analysis: A Catalog of Temptations and Pitfalls for Software Engineering Researchers","zh_title":"AI辅助定性数据分析中的反模式：软件工程研究者的诱惑与陷阱目录","primary_category":"cs.SE","date":"2026-08-31","score":2,"bucket":"other","tags":["AI辅助分析","定性研究","方法论"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.27927","has_summary":false},{"id":"2608.28420","title":"Between Algorithm (AI) and Intuition (Human): Preserving Designer Agency in AI-Assisted Sensemaking of Qualitative UX Data","zh_title":"算法（AI）与直觉（人类）之间：在AI辅助的定性用户体验数据意义建构中保持设计师能动性","primary_category":"cs.HC","date":"2026-08-31","score":2,"bucket":"other","tags":["AI辅助分析","设计研究","人机协作"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28420","has_summary":false},{"id":"2608.25553","title":"When Stale Constraints Go Unchecked: Budgeted Verification Failures in Inherited Agent Memory","zh_title":"当陈旧约束未被检查：继承智能体记忆中的预算验证失败","primary_category":"cs.IR","date":"2026-08-31","score":0,"bucket":"other","tags":["多智能体系统","记忆验证","约束一致性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.25553","has_summary":false},{"id":"2608.26159","title":"Self-Generated Text Recognition: Quality Heuristics, Cross-Task Transfer, and Downstream Bias in LLM Evaluation","zh_title":"自生成文本识别：LLM评估中的质量启发式、跨任务迁移与下游偏差","primary_category":"cs.CL","date":"2026-08-31","score":0,"bucket":"other","tags":["LLM自识别","模型评估","安全风险"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26159","has_summary":false},{"id":"2608.27638","title":"Generative AI Expands the Intellectual Reach of Course Based Undergraduate Research Experiences (CUREs)","zh_title":"生成式人工智能拓展了基于课程的本科生研究经验（CUREs）的智力范围","primary_category":"cs.AI","date":"2026-08-31","score":0,"bucket":"other","tags":["生成式AI","教育技术","本科生研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.27638","has_summary":false},{"id":"2608.27953","title":"The Illusion of $\\textit{What If}$: Evaluating the Breakdown of Counterfactual Reasoning in LLMs","zh_title":"《如果》的幻觉：评估大语言模型中反事实推理的崩溃","primary_category":"cs.AI","date":"2026-08-31","score":0,"bucket":"other","tags":["反事实推理","基准测试","因果推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.27953","has_summary":false},{"id":"2608.27629","title":"LitCurate: A Configuration-Driven AI-Assisted Framework for Scientific Database Construction with an Application to Lower-Mantle Equation-of-State Data","zh_title":"LitCurate：配置驱动的AI辅助科学数据库构建框架及其在下地幔状态方程数据中的应用","primary_category":"cs.IR","date":"2026-08-31","score":0,"bucket":"other","tags":["文献挖掘","数据库构建","LLM应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.27629","has_summary":false},{"id":"2608.27932","title":"Graphionale: How Graph Visualizations of LLM Rationales Affect Human Decision Making","zh_title":"Graphionale：LLM推理的图形可视化如何影响人类决策","primary_category":"cs.HC","date":"2026-08-31","score":0,"bucket":"other","tags":["人机交互","可解释AI","决策支持"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.27932","has_summary":false},{"id":"2608.28050","title":"Too Much of the Same: From Algorithmic to Human Bias in Learning to Defer","zh_title":"过多相同：从算法偏差到人类偏差在学习推迟中","primary_category":"cs.HC","date":"2026-08-31","score":0,"bucket":"other","tags":["人机协作","学习推迟","人类偏差"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.28050","has_summary":false},{"id":"2608.28501","title":"A Guided Inquiry Approach to Students Co-Designing Generative AI Course Policies","zh_title":"学生共同设计生成式AI课程政策的引导式探究方法","primary_category":"cs.CY","date":"2026-08-31","score":0,"bucket":"other","tags":["教育政策","生成式AI","学生参与"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.28501","has_summary":false},{"id":"2608.27538","title":"Disaffection at Work: Employee Responses to Job-Related Information","zh_title":"工作不满：员工对工作相关信息的反应","primary_category":"econ.GN","date":"2026-08-31","score":0,"bucket":"other","tags":["劳动经济学","随机调查实验","员工行为"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.27538","has_summary":false},{"id":"2608.26291","title":"Assessing mentalization in humans and large language models","zh_title":"评估人类与大语言模型的心理化能力","primary_category":"cs.AI","date":"2026-08-28","score":10,"bucket":"selected","tags":["LLM仿真","经济学实验","认知建模"],"rubric_hits":["A1","A2","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2608.26291","has_summary":true},{"id":"2608.26327","title":"How Unlikely Is \"Unlikely\"? Assessing Verbal Probability Perception Across Large Language Models","zh_title":"“不太可能”有多不可能？跨大语言模型评估言语概率感知","primary_category":"cs.CL","date":"2026-08-28","score":8,"bucket":"selected","tags":["LLM仿真","概率语言","人类基准对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.26327","has_summary":true},{"id":"2608.26188","title":"Is Your Neighborhood Safe? Place-based Stigma in Large Language Models' Urban Safety Judgments","zh_title":"你的社区安全吗？大语言模型城市安全判断中的地方污名","primary_category":"cs.AI","date":"2026-08-28","score":8,"bucket":"selected","tags":["LLM仿真","社会偏见","城市安全"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.26188","has_summary":true},{"id":"2608.26221","title":"Prompt Sensitivity of Generative Agents: Evidence from an Epidemic Model","zh_title":"生成式智能体的提示敏感性：来自流行病模型的证据","primary_category":"physics.soc-ph","date":"2026-08-28","score":8,"bucket":"selected","tags":["LLM仿真","流行病模型","提示敏感性"],"rubric_hits":["A1","A3","B4"],"abs_url":"https://arxiv.org/abs/2608.26221","has_summary":true},{"id":"2608.23705","title":"The Limits of Automatic Evaluation of Creativity in Large Language Models","zh_title":"大语言模型创造力自动评估的局限性","primary_category":"cs.CL","date":"2026-08-28","score":7,"bucket":"pending","tags":["LLM评估偏差","人类对照","创造力测量"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23705","has_summary":true},{"id":"2608.23780","title":"When Youth Enter The Chat: An Epistemic Shift in the Validation of LLM-Based Measures of Student Talk","zh_title":"当青少年进入聊天：基于LLM的学生话语测量验证的认识论转变","primary_category":"cs.CL","date":"2026-08-28","score":7,"bucket":"pending","tags":["LLM测量效度","批判性评估","教育话语分析"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.23780","has_summary":true},{"id":"2608.26899","title":"Counterfactual Bias Testing for Application Tracking System","zh_title":"申请追踪系统的反事实偏见测试","primary_category":"cs.AI","date":"2026-08-28","score":7,"bucket":"pending","tags":["LLM仿真","算法审计","公平性评估"],"rubric_hits":["A1","A2","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2608.26899","has_summary":true},{"id":"2608.27219","title":"BALMS: Benchmarking Agentic LLMs for Longitudinal Mental Health Sensing","zh_title":"BALMS：面向纵向心理健康感知的智能体LLM基准测试","primary_category":"cs.CL","date":"2026-08-28","score":6,"bucket":"other","tags":["LLM基准","心理健康","智能体"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.27219","has_summary":false},{"id":"2608.22444","title":"Aligned Alone, Misaligned Together: Forecasting Adversarial Capture in LLM Agent Populations","zh_title":"单独对齐，群体失准：预测 LLM 智能体群体中的对抗性俘获","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["多智能体模拟","安全评估","群体行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22444","has_summary":false},{"id":"2608.26154","title":"Evaluating AI Generated Summaries for Cancer Patients","zh_title":"评估面向癌症患者的AI生成摘要","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["LLM评估","医疗摘要","人类对照"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.26154","has_summary":false},{"id":"2608.26372","title":"Knowledge-Verified Emergent Deception in LLM Agents Under Conflicting Incentives","zh_title":"冲突激励下LLM智能体的知识验证涌现欺骗","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["LLM欺骗","智能体诚实性","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.26372","has_summary":false},{"id":"2608.26529","title":"Multi-Expert Conformal Risk Control for Pairwise LLM Judging in Open-Ended Dialogue","zh_title":"面向开放域对话中成对LLM评判的多专家共形风险控制","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["LLM评判","共形风险控制","多专家聚合"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.26529","has_summary":false},{"id":"2608.26674","title":"Do LLMs Understand Personality? Rethinking Persona Fidelity Evaluation through Structured Behavioral Inference","zh_title":"LLM理解人格吗？通过结构化行为推理重新思考人格保真度评估","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["人格保真度","LLM评估","心理测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.26674","has_summary":false},{"id":"2608.27402","title":"How Language Models Organize and Structure Moral Knowledge","zh_title":"语言模型如何组织与构建道德知识","primary_category":"cs.CL","date":"2026-08-28","score":5,"bucket":"other","tags":["道德知识表征","模型可解释性","道德基础理论"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.27402","has_summary":false},{"id":"2608.26150","title":"Leveraging Large Language Models for Systematic Literature Review of Disease Spread Models","zh_title":"利用大语言模型进行疾病传播模型的系统文献综述","primary_category":"cs.AI","date":"2026-08-28","score":5,"bucket":"other","tags":["LLM标注","系统综述","自动化"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.26150","has_summary":false},{"id":"2608.26885","title":"Evaluating human and LLM screening workflows in a conceptually complex scoping review: Recall--workload trade-offs and run-to-run consistency","zh_title":"在概念复杂的范围综述中评估人类与LLM筛选工作流：召回率-工作量权衡与运行间一致性","primary_category":"cs.AI","date":"2026-08-28","score":5,"bucket":"other","tags":["LLM辅助筛选","人机对比","证据综合"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.26885","has_summary":false},{"id":"2608.27443","title":"Do User-Authored Permission Policies Improve Protection Against AI Agent Overreach?","zh_title":"用户编写的权限策略能否改善对AI代理越权行为的防护？","primary_category":"cs.HC","date":"2026-08-28","score":5,"bucket":"other","tags":["AI代理","权限策略","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.27443","has_summary":false},{"id":"2607.27747","title":"Can LVLMs Uncover the Truth Behind Visual Illusions? An Analysis of Perceptual and Reasoning Capabilities","zh_title":"LVLM能否揭示视觉错觉背后的真相？感知与推理能力分析","primary_category":"cs.CL","date":"2026-08-28","score":2,"bucket":"other","tags":["视觉语言模型","基准评测","视觉错觉"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.27747","has_summary":false},{"id":"2608.01942","title":"CultureVidBench: Benchmarking Cultural Understanding in Text-to-Video Generation","zh_title":"CultureVidBench：文本到视频生成中文化理解的基准测试","primary_category":"cs.CV","date":"2026-08-28","score":2,"bucket":"other","tags":["文本到视频生成","文化理解","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.01942","has_summary":false},{"id":"2608.26131","title":"Evaluating Language Models in Realistic Conversational Contexts","zh_title":"在真实对话情境中评估语言模型","primary_category":"cs.CL","date":"2026-08-28","score":2,"bucket":"other","tags":["对话评测","基准数据集","LLM评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26131","has_summary":false},{"id":"2608.26137","title":"Interpretable, Fairly Evaluated Automated L2 Speaking Assessment that Beats the Single-Human Ceiling and Why Pause Encoding Does Not Change LLM Fluency Scores","zh_title":"可解释、公平评估的自动化二语口语评分超越单人上限及停顿编码为何不改变LLM流利度分数","primary_category":"cs.CL","date":"2026-08-28","score":2,"bucket":"other","tags":["自动口语评分","LLM评估","二语习得"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26137","has_summary":false},{"id":"2608.26511","title":"Sycophancy Suppression Can Impair Rational Updating: Anti-Sycophancy Should Preserve the Ability to Update","zh_title":"抑制谄媚可能损害理性更新：反谄媚应保留更新能力","primary_category":"cs.CL","date":"2026-08-28","score":2,"bucket":"other","tags":["LLM行为","谄媚","模型对齐"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26511","has_summary":false},{"id":"2608.27049","title":"Research Design Tracking and Assessment for the Social Sciences","zh_title":"社会科学研究设计追踪与评估","primary_category":"cs.CL","date":"2026-08-28","score":2,"bucket":"other","tags":["研究设计评估","LLM评测","因果推断"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.27049","has_summary":false},{"id":"2608.26958","title":"Scaling Model-Generated Distillation Data Can Make Latent Teacher Traits More Recoverable","zh_title":"扩展模型生成蒸馏数据可使潜在教师特质更易恢复","primary_category":"cs.LG","date":"2026-08-28","score":2,"bucket":"other","tags":["模型蒸馏","特质传递","数据扩展"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26958","has_summary":false},{"id":"2608.27364","title":"Sophistication in GenAI Use: Field Evidence from a Large Firm","zh_title":"生成式AI使用的复杂性：来自一家大型企业的实地证据","primary_category":"cs.AI","date":"2026-08-28","score":2,"bucket":"other","tags":["GenAI使用","企业实地研究","员工行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27364","has_summary":false},{"id":"2608.26135","title":"Data Science Approaches to Evaluating Honours Candidates","zh_title":"评估荣誉候选人的数据科学方法","primary_category":"cs.CL","date":"2026-08-28","score":0,"bucket":"other","tags":["情感分析","OSINT","NLP管道"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26135","has_summary":false},{"id":"2608.26161","title":"Mutual Debiasing via Dual-Seed Comparison for Probabilistic Sampling in Large Language Models","zh_title":"基于双种子比较的互去偏方法用于大语言模型概率采样","primary_category":"cs.CL","date":"2026-08-28","score":0,"bucket":"other","tags":["概率采样","偏差校正","LLM生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26161","has_summary":false},{"id":"2608.26887","title":"Planting a Latent Variable in Natural-Looking Text: a More Realistic Test of Belief States in LLMs and Their Link to Concept Geometry","zh_title":"在自然文本中植入潜变量：对LLM信念状态及其与概念几何联系的更真实测试","primary_category":"cs.CL","date":"2026-08-28","score":0,"bucket":"other","tags":["LLM信念状态","概念几何","潜变量"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26887","has_summary":false},{"id":"2608.26226","title":"LLM Agents for Time-Series: A Survey","zh_title":"面向时间序列的LLM智能体：综述","primary_category":"cs.AI","date":"2026-08-28","score":0,"bucket":"other","tags":["LLM智能体","时间序列","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26226","has_summary":false},{"id":"2608.26753","title":"Beyond Execution: Auditing Experimental Fidelity in LLM-Driven Scientific Research","zh_title":"超越执行：审计LLM驱动科学研究中的实验保真度","primary_category":"cs.SE","date":"2026-08-28","score":0,"bucket":"other","tags":["AI科学家","实验保真度","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26753","has_summary":false},{"id":"2608.27072","title":"Emotional Preferences as Goal-Priority Regulation","zh_title":"情感偏好作为目标优先级调节","primary_category":"cs.LG","date":"2026-08-28","score":0,"bucket":"other","tags":["强化学习","多目标决策","情感计算"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27072","has_summary":false},{"id":"2608.27238","title":"Assessing Company Contributions to Societal Resilience: Extending the Societal Capacity Assessment Framework to Agentic AI","zh_title":"评估公司对社会韧性的贡献：将社会能力评估框架扩展到代理型AI","primary_category":"cs.CY","date":"2026-08-28","score":0,"bucket":"other","tags":["AI治理","社会韧性","公司责任"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.27238","has_summary":false},{"id":"2608.26797","title":"On the Indistinguishability of Human v/s AI Generated Text","zh_title":"论人类与AI生成文本的不可区分性","primary_category":"cs.LG","date":"2026-08-28","score":0,"bucket":"other","tags":["AI文本检测","文本改写","对抗性样本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.26797","has_summary":false},{"id":"2608.26388","title":"Assessing Socio-Cyber Vulnerability Using Survey and Social Media Data","zh_title":"利用调查和社交媒体数据评估社会网络脆弱性","primary_category":"cs.SI","date":"2026-08-28","score":0,"bucket":"other","tags":["社会网络脆弱性","网络安全","人类数据"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.26388","has_summary":false},{"id":"2608.26358","title":"An Anonymized Urn-Based Experimental Dataset on Decision-Making under Risk and Ambiguity","zh_title":"风险与模糊决策的匿名瓮实验数据集","primary_category":"econ.GN","date":"2026-08-28","score":0,"bucket":"other","tags":["实验数据集","风险决策","模糊决策"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.26358","has_summary":false},{"id":"2608.24912","title":"Analyzing and Correcting Benevolence Bias in Large Language Models","zh_title":"分析和纠正大语言模型中的仁慈偏差","primary_category":"cs.HC","date":"2026-08-27","score":10,"bucket":"selected","tags":["LLM仿真","算法保真度","偏差校正"],"rubric_hits":["A1","A2","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.24912","has_summary":true},{"id":"2604.06223","title":"The Quiet and the Compliant: How Regulation and Polarization Shape Conventional Wisdoms on Corporate Social Engagement in High-risk Settings","zh_title":"沉默与顺从：监管与极化如何塑造高风险环境下企业社会参与的常规智慧","primary_category":"physics.soc-ph","date":"2026-08-27","score":9,"bucket":"selected","tags":["LLM仿真","合成调查","企业社会责任"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2604.06223","has_summary":true},{"id":"2608.25771","title":"Large Language Model Few-Shot Prompting with Dilemma Training Outperforms Human Surrogates in Predicting Patient Preferences","zh_title":"基于困境训练的大语言模型少样本提示在预测患者偏好上超越人类代理","primary_category":"cs.HC","date":"2026-08-27","score":9,"bucket":"selected","tags":["LLM仿真","患者偏好预测","人类对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.25771","has_summary":true},{"id":"2608.24920","title":"Semantic Variability of Replies Across LLMs: Implications for Designing Conversation-Based Assessment","zh_title":"不同大语言模型回复的语义变异性：对设计基于对话的评估的启示","primary_category":"cs.CL","date":"2026-08-27","score":7,"bucket":"pending","tags":["LLM仿真","语义一致性","对话评估"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.24920","has_summary":true},{"id":"2608.25999","title":"Distinct dynamics of conceptual and referential disruptions in human reading and large language model processing","zh_title":"人类阅读与大语言模型处理中概念与指称干扰的不同动态","primary_category":"cs.CL","date":"2026-08-27","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","认知建模"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.25999","has_summary":true},{"id":"2608.25236","title":"Rare Diseases, Common Dilemmas: LLMs Prioritize Equal Resource Distribution over Patient Benefit in Decision-Making","zh_title":"罕见病，常见困境：LLM在决策中优先考虑资源平等分配而非患者获益","primary_category":"cs.CY","date":"2026-08-27","score":7,"bucket":"pending","tags":["LLM决策仿真","伦理决策","资源分配"],"rubric_hits":["A1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.25236","has_summary":true},{"id":"2608.24908","title":"Hallucination by proxy in LLM-assisted differential diagnosis","zh_title":"LLM辅助鉴别诊断中的代理幻觉","primary_category":"cs.HC","date":"2026-08-27","score":7,"bucket":"pending","tags":["LLM幻觉","医生决策","人类行为对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.24908","has_summary":true},{"id":"2607.22188","title":"Draining the Energy Commons: Self-Defeating Over-Appropriation as a Coordination Failure in Agentic LLM Collectives","zh_title":"耗尽能源公地：智能体 LLM 集体中作为协调失败的自我挫败式过度占用","primary_category":"cs.MA","date":"2026-08-27","score":6,"bucket":"other","tags":["LLM 多智能体","公共资源困境","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.22188","has_summary":true},{"id":"2608.24076","title":"AgentWorld: Personality-Aware Reliability Evaluation for Agentic Information Retrieval","zh_title":"AgentWorld：面向智能体信息检索的人格感知可靠性评估","primary_category":"cs.AI","date":"2026-08-27","score":6,"bucket":"other","tags":["人格驱动仿真","智能体评估","可靠性测试"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2608.24076","has_summary":false},{"id":"2608.25152","title":"Belief Cascades Drive Persuasion in LLM Agent Networks","zh_title":"信念级联驱动 LLM 智能体网络中的说服","primary_category":"cs.CL","date":"2026-08-27","score":6,"bucket":"other","tags":["LLM 多智能体","说服动态","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.25152","has_summary":false},{"id":"2608.24896","title":"Agentic World Analysis (AWA) - an alternative way to explore systems and support decision making","zh_title":"智能体世界分析（AWA）——探索系统和支持决策的另一种方式","primary_category":"cs.HC","date":"2026-08-27","score":6,"bucket":"other","tags":["LLM社会模拟","情景分析","决策支持"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.24896","has_summary":false},{"id":"2608.25623","title":"Using profiles of cognitive capability to assess AI suitability for workplace tasks","zh_title":"利用认知能力画像评估AI在工作任务中的适用性","primary_category":"cs.AI","date":"2026-08-27","score":6,"bucket":"other","tags":["AI任务分配","认知能力画像","人机协作"],"rubric_hits":["D1","D2"],"abs_url":"https://arxiv.org/abs/2608.25623","has_summary":false},{"id":"2608.24001","title":"Diverse by Reasoning: Harnessing the Wisdom of LLM Crowds for Future Prediction","zh_title":"通过推理实现多样性：利用LLM群体智慧进行未来预测","primary_category":"cs.AI","date":"2026-08-27","score":5,"bucket":"other","tags":["LLM群体智慧","未来预测","行为多样性"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.24001","has_summary":false},{"id":"2608.25824","title":"Localize-Then-Decide Guarantees for LLM Judgments","zh_title":"LLM判断的局部化后决策保证","primary_category":"cs.CL","date":"2026-08-27","score":5,"bucket":"other","tags":["LLM评估","校准","人类判断对齐"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.25824","has_summary":false},{"id":"2608.25854","title":"Key Point Analysis Needs Structure Recovery: Task Definition, Dataset Diagnosis, and a Structure-Aware Benchmark","zh_title":"关键点分析需要结构恢复：任务定义、数据集诊断与结构感知基准","primary_category":"cs.CL","date":"2026-08-27","score":5,"bucket":"other","tags":["关键点分析","LLM评估","数据集标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.25854","has_summary":false},{"id":"2608.26081","title":"SwarmWorld: Stigmergic technological evolution in societies of language-model agents","zh_title":"SwarmWorld：语言模型智能体社会中的共识主动性技术演化","primary_category":"cs.AI","date":"2026-08-27","score":5,"bucket":"other","tags":["多智能体系统","社会模拟","集体智能"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.26081","has_summary":false},{"id":"2608.24903","title":"Evidence-Grounded Mapping of Multimodal Human Sensing Psychological Transdiagnostic Dimensions","zh_title":"基于证据的多模态人类感知心理跨诊断维度映射","primary_category":"cs.HC","date":"2026-08-27","score":5,"bucket":"other","tags":["LLM标注","心理健康","多模态数据"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24903","has_summary":false},{"id":"2608.25267","title":"Mitigating LLM sycophancy with RL-based fine-tuning: Bayesian Truth Serum approach","zh_title":"基于强化学习的微调缓解大语言模型谄媚性：贝叶斯真值血清方法","primary_category":"cs.LG","date":"2026-08-27","score":5,"bucket":"other","tags":["LLM谄媚性","强化学习微调","模型行为校准"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.25267","has_summary":false},{"id":"2608.25832","title":"Skill Issue: Are Skills Language-Invariant in LLMs?","zh_title":"技能问题：LLM的技能是否跨语言不变？","primary_category":"cs.CL","date":"2026-08-27","score":3,"bucket":"other","tags":["多智能体","跨语言评估","文本游戏"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.25832","has_summary":false},{"id":"2608.01666","title":"Style Wins, Substance Loses: A Diagnosis of LLM-as-Judge in Idea Generation","zh_title":"风格胜，实质败：LLM作为评审在创意生成中的诊断","primary_category":"cs.CL","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM评审","文体偏差","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01666","has_summary":false},{"id":"2608.24901","title":"Detection != Reliable Control: Decodable Empathy Directions Yield at Most Partial Shifts in Automated Empathy Scores","zh_title":"检测不等于可靠控制：可解码的共情方向至多导致自动共情分数的部分偏移","primary_category":"cs.CL","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM可控性","共情方向","模型行为分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.24901","has_summary":false},{"id":"2608.25654","title":"Unmatched Does Not Mean False: Incomplete Reference Sets Can Reverse Calibration Rankings in Open-Ended Theory-of-Mind Tracking","zh_title":"未匹配不等于错误：不完整参考集可逆转开放式心智理论追踪中的校准排名","primary_category":"cs.CL","date":"2026-08-27","score":2,"bucket":"other","tags":["心智理论","校准评估","自然语言处理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25654","has_summary":false},{"id":"2608.25660","title":"Think-Probe-Respond: Improving Large Language Models as Judges of Research Idea Novelty","zh_title":"思考-探测-回应：改进大语言模型作为研究想法新颖性判断器","primary_category":"cs.CL","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM评测","新颖性判断","校准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25660","has_summary":false},{"id":"2608.25869","title":"Anchoring Bias in LLM-as-a-Judge Systems: Prior Scores Compromise Evaluation Independence","zh_title":"LLM作为评判者系统中的锚定偏差：先前分数损害评估独立性","primary_category":"cs.CL","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM评估","锚定偏差","评判者偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25869","has_summary":false},{"id":"2608.25325","title":"FinRiskAtlas: Decision-Aligned Evaluation of Large Language Models for Financial Risk Review","zh_title":"FinRiskAtlas：面向金融风险审查的大语言模型决策对齐评估","primary_category":"cs.AI","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM评测","金融风险审查","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25325","has_summary":false},{"id":"2608.24899","title":"aipsy-judge: A Specialized, Psychologist-Corrected Local Judge for the Psychological Safety of Conversational AI","zh_title":"aipsy-judge：面向对话式AI心理安全的专用、经心理学家校正的本地裁判模型","primary_category":"cs.HC","date":"2026-08-27","score":2,"bucket":"other","tags":["LLM裁判","心理安全","对话AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.24899","has_summary":false},{"id":"2608.24900","title":"Stronger Alignment between Brain Activity and LLM Embeddings during Code Writing compared to Prose Writing","zh_title":"代码写作相比散文写作时大脑活动与LLM嵌入的对齐更强","primary_category":"cs.HC","date":"2026-08-27","score":2,"bucket":"other","tags":["脑机对齐","神经编码","LLM表征"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24900","has_summary":false},{"id":"2608.24914","title":"AI-Ready Research Workflows in Computational Social Science: Lessons on Building a Shared Language for Interdisciplinary Collaboration","zh_title":"计算社会科学中的AI就绪研究工作流：跨学科协作共享语言构建的经验教训","primary_category":"cs.HC","date":"2026-08-27","score":2,"bucket":"other","tags":["AI工作流","计算社会科学","LLM分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24914","has_summary":false},{"id":"2608.25316","title":"AVI-Personality: A Trait-Activated Multimodal Dataset for Personality and Competency Assessment in Asynchronous Video Interviews","zh_title":"AVI-Personality：异步视频面试中人格与胜任力评估的特质激活多模态数据集","primary_category":"cs.HC","date":"2026-08-27","score":2,"bucket":"other","tags":["人格评估","多模态数据集","视频面试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25316","has_summary":false},{"id":"2608.25871","title":"CEDAR: Controlled and Event-Driven Demand Forecasting via Residual Decomposition","zh_title":"CEDAR：通过残差分解实现受控且事件驱动的需求预测","primary_category":"cs.LG","date":"2026-08-27","score":2,"bucket":"other","tags":["时间序列预测","电商需求预测","LLM辅助表示"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.25871","has_summary":false},{"id":"2606.07392","title":"Online Pandora's Box for Contextual LLM Cascading","zh_title":"面向上下文LLM级联的在线潘多拉魔盒","primary_category":"cs.AI","date":"2026-08-27","score":0,"bucket":"other","tags":["LLM级联","在线学习","决策优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.07392","has_summary":false},{"id":"2608.00991","title":"SCHEDBench: A Benchmark for Evaluating LLM Constraint Faithfulness in Natural-Language Combinatorial Scheduling","zh_title":"SCHEDBench：评估LLM在自然语言组合调度中约束忠实度的基准","primary_category":"cs.AI","date":"2026-08-27","score":0,"bucket":"other","tags":["LLM评测","调度问题","约束遵循"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00991","has_summary":false},{"id":"2608.24662","title":"The Invisible Editorial Layer: Formalizing Undisclosed Inference-Time Steering, Probability Placement, and the Attribution Problem in Deployed Language Models","zh_title":"隐形编辑层：形式化部署语言模型中未公开的推理时引导、概率放置与归因问题","primary_category":"cs.AI","date":"2026-08-27","score":0,"bucket":"other","tags":["推理时干预","模型治理","归因问题"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24662","has_summary":false},{"id":"2608.24952","title":"The Dialect Tax: Dialectal Biases Persist throughout the Language Modeling Pipeline","zh_title":"方言税：方言偏见贯穿语言建模全流程","primary_category":"cs.CL","date":"2026-08-27","score":0,"bucket":"other","tags":["方言偏见","语言模型公平性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24952","has_summary":false},{"id":"2608.25717","title":"When RAG Fails to Equalize: Geo-bias in Factual Question Answering over Public Companies","zh_title":"当RAG无法实现均衡：上市公司事实问答中的地理偏差","primary_category":"cs.CL","date":"2026-08-27","score":0,"bucket":"other","tags":["RAG","地理偏差","事实问答"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.25717","has_summary":false},{"id":"2608.26089","title":"From Producing to Validating: How AI Is Deskilling Freelancers","zh_title":"从生产到验证：AI如何使自由职业者去技能化","primary_category":"cs.HC","date":"2026-08-27","score":0,"bucket":"other","tags":["AI对劳动力影响","自由职业","去技能化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26089","has_summary":false},{"id":"2608.25063","title":"The AI Adaptation Gap in Higher Education: Students, Faculty, and Administrative Staff","zh_title":"高等教育中的AI适应差距：学生、教师与行政人员","primary_category":"cs.CY","date":"2026-08-27","score":0,"bucket":"other","tags":["AI使用调查","高等教育","人类态度"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.25063","has_summary":false},{"id":"2608.26075","title":"Epistemic Networks, Collective Misperception, and the Manipulation of Social Knowledge","zh_title":"认知网络、集体误解与社会知识的操纵","primary_category":"cs.SI","date":"2026-08-27","score":0,"bucket":"other","tags":["社会网络","多智能体系统","理论模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.26075","has_summary":false},{"id":"2608.25678","title":"Normative boundaries of AI in scientific work: Evidence from PhD researchers","zh_title":"人工智能在科学工作中的规范边界：来自博士研究生的证据","primary_category":"econ.GN","date":"2026-08-27","score":0,"bucket":"other","tags":["AI态度","科学工作","调查分析"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.25678","has_summary":false},{"id":"2608.20539","title":"ExploraTwin, a Non-Profit Research Platform for Digital Twin Simulations","zh_title":"ExploraTwin：一个用于数字孪生仿真的非营利研究平台","primary_category":"cs.CY","date":"2026-08-26","score":9,"bucket":"selected","tags":["数字孪生","调查仿真","平台"],"rubric_hits":["A1","A2","A4","B1"],"abs_url":"https://arxiv.org/abs/2608.20539","has_summary":true},{"id":"2608.21296","title":"Level-k Distinguishable Mechanisms for Evaluating Bounded Rationality in LLMs","zh_title":"评估LLM有限理性的Level-k可区分机制","primary_category":"cs.MA","date":"2026-08-26","score":7,"bucket":"pending","tags":["LLM策略推理","博弈实验","有限理性"],"rubric_hits":["A1","B3"],"abs_url":"https://arxiv.org/abs/2608.21296","has_summary":true},{"id":"2608.23818","title":"Beyond Static and Linear: What Attention Constraints Best Fit Human Reading Times?","zh_title":"超越静态与线性：何种注意力约束最拟合人类阅读时间？","primary_category":"cs.CL","date":"2026-08-26","score":7,"bucket":"pending","tags":["认知建模","注意力约束","心理测量拟合"],"rubric_hits":["A2","B1"],"abs_url":"https://arxiv.org/abs/2608.23818","has_summary":true},{"id":"2608.23640","title":"Auditing the Synthetic Memoir: Measuring Scene-Level Confabulation in LLM-Generated Autobiography Against the Documented Record of the Life It Describes","zh_title":"审计合成回忆录：对照真实生活记录测量LLM生成自传中的场景级虚构","primary_category":"cs.AI","date":"2026-08-26","score":7,"bucket":"pending","tags":["LLM仿真","真实性审计","偏差评估"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23640","has_summary":true},{"id":"2608.23906","title":"Quantifying System-Level Harms from AI Adoption in Complex Sociotechnical Systems","zh_title":"量化复杂社会技术系统中AI采纳的系统级危害","primary_category":"cs.AI","date":"2026-08-26","score":7,"bucket":"pending","tags":["LLM仿真","金融系统","系统性风险"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.23906","has_summary":true},{"id":"2608.24046","title":"Algorithmic Impact Reveals the Hidden Social Choice Structure of Alignment","zh_title":"算法影响揭示对齐的隐藏社会选择结构","primary_category":"cs.AI","date":"2026-08-26","score":7,"bucket":"pending","tags":["LLM对齐","社会选择","人类偏好"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.24046","has_summary":true},{"id":"2608.23966","title":"Who Chooses How Preferences Are Aggregated? Auditing Aggregation-Rule Authority in LLM-Based Group Recommendation","zh_title":"谁选择偏好如何聚合？审计基于LLM的群体推荐中的聚合规则权威","primary_category":"cs.HC","date":"2026-08-26","score":7,"bucket":"pending","tags":["LLM仿真","群体决策","偏好聚合"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.23966","has_summary":true},{"id":"2507.21790","title":"Can large language models assist choice modelling? Insights into prompting strategies and current models' capabilities","zh_title":"大语言模型能否辅助选择建模？对提示策略和当前模型能力的洞察","primary_category":"econ.EM","date":"2026-08-26","score":6,"bucket":"other","tags":["LLM辅助建模","离散选择模型","提示策略"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2507.21790","has_summary":true},{"id":"2608.19551","title":"Delegating or Doing? Understanding User Behavior in Hybrid Human-Agent Interfaces","zh_title":"委托还是亲为？理解混合人机交互界面中的用户行为","primary_category":"cs.HC","date":"2026-08-26","score":5,"bucket":"other","tags":["人机交互","LLM代理","用户行为"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.19551","has_summary":false},{"id":"2608.20373","title":"An ambiguity taxonomy for evaluating large language model performance on clinical registry abstraction: a multi-site prospective study","zh_title":"评估大语言模型在临床注册数据提取中表现的不确定性分类法：一项多中心前瞻性研究","primary_category":"cs.CL","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM标注","临床数据提取","不确定性分类"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.20373","has_summary":false},{"id":"2608.23766","title":"What Reaches Expert Review? Representation, Structural Screening, and Candidate-Form Dependence in AI-Assisted Item Development","zh_title":"什么能到达专家评审？AI辅助条目开发中的表征、结构筛选与候选形式依赖","primary_category":"cs.CL","date":"2026-08-26","score":5,"bucket":"other","tags":["AI辅助条目生成","心理测量","计算评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.23766","has_summary":false},{"id":"2608.23641","title":"How much of a measured AI preference is the model, and how much is the instrument?","zh_title":"测得的AI偏好中，多少来自模型本身，多少来自测量工具？","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["模型偏好测量","测量工具效度","AI福利"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.23641","has_summary":false},{"id":"2608.23837","title":"SyPS: Measuring Sycophancy Prompt Sensitivity in Large Language Models","zh_title":"SyPS：测量大语言模型中的谄媚提示敏感性","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM社会行为","谄媚测量","提示敏感性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.23837","has_summary":false},{"id":"2608.23979","title":"Rules Before Oracles: Auditable, User-Configurable Argument Selection for Deliberative Polling","zh_title":"规则先于神谕：面向协商式民调的可审计、用户可配置论证选择","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM社会模拟","协商民主","可审计推荐"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.23979","has_summary":false},{"id":"2608.24419","title":"A Judge Should Know What Changed:Construct Validity for LLM-as-a-Judge Evaluation","zh_title":"评判者应知何所变：LLM作为评判者评估的构念效度","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM评估","构念效度","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24419","has_summary":false},{"id":"2608.24545","title":"Discovering Adaptive Transmission Programs for Collective Innovation","zh_title":"发现集体创新的自适应传输程序","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM多智能体","社会模拟","集体智能"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.24545","has_summary":false},{"id":"2608.24825","title":"A Dual-Dimensional LLM Framework for Automated Item Incidental Content Similarity Analysis in Large-Scale Assessments","zh_title":"用于大规模评估中项目附带内容相似性分析的双维LLM框架","primary_category":"cs.AI","date":"2026-08-26","score":5,"bucket":"other","tags":["LLM标注","项目相似性","自适应测试"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24825","has_summary":false},{"id":"2608.24297","title":"When AI \"Works,\" When Does Help Begin?: Intergenerational Support Around Older Adults' LLM Usage","zh_title":"当AI“有效”时，帮助何时开始？：围绕老年人LLM使用的代际支持","primary_category":"cs.HC","date":"2026-08-26","score":5,"bucket":"other","tags":["人机交互","老年人技术使用","代际支持"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.24297","has_summary":false},{"id":"2608.24215","title":"Agentopia on a Consumer GPU: A Reduced-Scale Long-Horizon Port with an 8B Model","zh_title":"消费级GPU上的Agentopia：基于8B模型的缩减规模长时程移植","primary_category":"cs.MA","date":"2026-08-26","score":5,"bucket":"other","tags":["多智能体社会模拟","LLM仿真","资源受限部署"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.24215","has_summary":false},{"id":"2608.23644","title":"Ethical LLM-Assisted Research: A Framework for Responsible Delegation, Verification, and Epistemic Value","zh_title":"伦理的LLM辅助研究：负责任委托、验证与认知价值的框架","primary_category":"cs.AI","date":"2026-08-26","score":4,"bucket":"other","tags":["科研伦理","LLM辅助研究","认知责任"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23644","has_summary":false},{"id":"2608.23050","title":"What Makes an Initial Reaction Ready for Discussion?: Multi-Persona AI Support for Stance Reflection and Writing","zh_title":"什么使初步反应准备好进行讨论？：多人格AI支持立场反思与写作","primary_category":"cs.HC","date":"2026-08-26","score":3,"bucket":"other","tags":["人机交互","立场反思","多角色AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.23050","has_summary":false},{"id":"2608.19760","title":"Credit Without Ground Truth: Auditing Step-Level Credit Assignment in LLM Agents Against Executed Replay","zh_title":"无真实基准的信用分配：基于执行重放的LLM智能体步骤级信用审计","primary_category":"cs.LG","date":"2026-08-26","score":2,"bucket":"other","tags":["LLM智能体","信用分配","工具使用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19760","has_summary":false},{"id":"2608.20405","title":"ARGUS: Theory-of-Mind Guided Argument Generation with Strategy-Aware Planning and Knowledge Grounding","zh_title":"ARGUS：基于心理理论引导的策略感知规划与知识支撑的论证生成","primary_category":"cs.CL","date":"2026-08-26","score":2,"bucket":"other","tags":["论证生成","心理理论","说服性写作"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.20405","has_summary":false},{"id":"2608.23271","title":"Expectations and Practices around AI Disclosure in CS Research","zh_title":"计算机科学研究中AI披露的期望与实践","primary_category":"cs.CY","date":"2026-08-26","score":2,"bucket":"other","tags":["AI披露","研究伦理","政策分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23271","has_summary":false},{"id":"2608.24189","title":"MemUse: Moving Memory Evaluation from Direct QA to Natural Integration in Long-Term Human-AI Conversation","zh_title":"MemUse：将记忆评估从直接问答转向长期人机对话中的自然整合","primary_category":"cs.CL","date":"2026-08-26","score":2,"bucket":"other","tags":["对话系统","记忆评估","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.24189","has_summary":false},{"id":"2608.24842","title":"Reading Is Not Using: Retrieval, Judgment, and the Design of AI Financial Research Workflows","zh_title":"阅读并非使用：检索、判断与AI金融研究工作流的设计","primary_category":"cs.CL","date":"2026-08-26","score":2,"bucket":"other","tags":["LLM金融分析","检索整合","工作流架构"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24842","has_summary":false},{"id":"2608.23622","title":"LLM Agents Perform Controlled Experiments Using Simulation Models","zh_title":"LLM智能体利用仿真模型进行受控实验","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["多智能体系统","仿真实验","过程优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23622","has_summary":false},{"id":"2608.23814","title":"Learning to Grade Efficiently: A Bandit-Driven Prompt-Selection Framework for Low-Cost LLM Essay Scoring","zh_title":"高效评分学习：面向低成本LLM作文评分的Bandit驱动提示选择框架","primary_category":"cs.LG","date":"2026-08-26","score":2,"bucket":"other","tags":["自动作文评分","多臂老虎机","提示选择"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23814","has_summary":false},{"id":"2608.23706","title":"Do LLMs Understand Limit Order Book Dynamics?","zh_title":"大语言模型理解限价订单簿动态吗？","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["LLM","限价订单簿","金融仿真"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23706","has_summary":false},{"id":"2608.23908","title":"Retrieval-augmented generation vs. deterministic tax computation in multi-agent financial advisory: A 2x2 factorial experiment","zh_title":"检索增强生成与确定性税务计算在多智能体金融咨询中的对比：一项2x2析因实验","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["多智能体系统","金融咨询","RAG"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23908","has_summary":false},{"id":"2608.23978","title":"When Seeing Is Not Enough: Benchmarking Interactive Visual Grounding in LVLMs","zh_title":"当看见还不够：基准测试大视觉语言模型中的交互式视觉定位","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["视觉定位","多模态模型","能力评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23978","has_summary":false},{"id":"2608.24069","title":"Poisoning Agentic Alpha: Adversarial Vulnerabilities Across Roles and Architectures in Multi-Agent Trading Systems","zh_title":"毒化智能体Alpha：多智能体交易系统中跨角色与架构的对抗性漏洞","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["多智能体系统","对抗攻击","金融交易"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24069","has_summary":false},{"id":"2608.24369","title":"Do Recipes Have Personas? Characterizing and Generating Creator Style in Attributed Procedural Graphs","zh_title":"食谱有人格吗？表征与生成归因程序图中的创作者风格","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["程序图生成","风格表征","LLM生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24369","has_summary":false},{"id":"2608.24570","title":"EviDx: Evidence-Aware Active Diagnosis with Scaffolded LLM Agents","zh_title":"EviDx：基于证据的主动诊断与脚手架式LLM智能体","primary_category":"cs.AI","date":"2026-08-26","score":2,"bucket":"other","tags":["多智能体系统","临床诊断","LLM智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24570","has_summary":false},{"id":"2608.23660","title":"From Causal Plausibility to Causal Reliability: Evaluating LLMs as Calibrated Direct Causal-Edge Classifiers","zh_title":"从因果合理性到因果可靠性：评估LLM作为校准的直接因果边分类器","primary_category":"cs.LG","date":"2026-08-26","score":2,"bucket":"other","tags":["因果发现","模型评测","置信度校准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23660","has_summary":false},{"id":"2608.23776","title":"Disentangled Skill Representations for Predictive Human Modeling","zh_title":"用于预测性人类建模的解耦技能表征","primary_category":"cs.LG","date":"2026-08-26","score":2,"bucket":"other","tags":["人类技能建模","表征学习","AI教练"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23776","has_summary":false},{"id":"2608.23567","title":"Whose Psychiatry Was Summoned? A Clinical Response to the Psychodynamic Assessment of Claude Mythos Preview","zh_title":"召唤了谁的精神病学？对Claude Mythos Preview心理动力学评估的临床回应","primary_category":"cs.CY","date":"2026-08-26","score":2,"bucket":"other","tags":["AI精神评估","临床回应","模型福利"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.23567","has_summary":false},{"id":"2608.23937","title":"An Echo Chamber of One: Should AI Psychosis Be a Distinct Clinical Entity?","zh_title":"一个人的回音室：AI精神病应成为独立临床实体吗？","primary_category":"cs.CY","date":"2026-08-26","score":2,"bucket":"other","tags":["AI精神病","聊天机器人","临床精神病学"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.23937","has_summary":false},{"id":"2608.23999","title":"The urban right to AI: Pluralistic co-design and governance of public space","zh_title":"城市AI权利：公共空间的多元共治与治理","primary_category":"cs.CY","date":"2026-08-26","score":2,"bucket":"other","tags":["城市AI治理","计算机视觉","参与式设计"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23999","has_summary":false},{"id":"2604.06621","title":"The Theorems of Dr. David Blackwell and Their Contributions to Artificial Intelligence","zh_title":"大卫·布莱克威尔博士的定理及其对人工智能的贡献","primary_category":"cs.GL","date":"2026-08-26","score":0,"bucket":"other","tags":["数学定理","人工智能综述","统计学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.06621","has_summary":false},{"id":"2608.01378","title":"When May a Model Replace the Experiment? Audits, Licenses, and the Price of Trust in Surrogate-Driven Design","zh_title":"模型何时能替代实验？代理驱动设计中的审计、许可与信任代价","primary_category":"cs.LG","date":"2026-08-26","score":0,"bucket":"other","tags":["机器学习代理","实验设计","可靠性审计"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.01378","has_summary":false},{"id":"2608.09696","title":"Model Discovery Agent: LLM-assisted Bayesian experiment design for data-efficient discovery of mechanistic world models","zh_title":"模型发现智能体：LLM辅助的贝叶斯实验设计用于数据高效的机制世界模型发现","primary_category":"cs.AI","date":"2026-08-26","score":0,"bucket":"other","tags":["科学发现","贝叶斯实验设计","LLM辅助建模"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09696","has_summary":false},{"id":"2608.16645","title":"Reconstruction: A Blind Benchmark for Recovering Research Ideas from Pre-Publication Bibliographies","zh_title":"重建：从发表前参考文献中恢复研究想法的盲测基准","primary_category":"cs.AI","date":"2026-08-26","score":0,"bucket":"other","tags":["基准测试","多智能体","科学发现"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16645","has_summary":false},{"id":"2608.23420","title":"Systematic Bias in Green Patent Classification: Silent Green and False Green","zh_title":"绿色专利分类中的系统性偏差：隐性绿色与虚假绿色","primary_category":"econ.EM","date":"2026-08-26","score":0,"bucket":"other","tags":["专利分类","LLM辅助标注","绿色技术"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23420","has_summary":false},{"id":"2608.23719","title":"ADE: Agentic Data Evolution Framework for Human-Centered Objectives","zh_title":"ADE：面向人类中心目标的智能体数据演化框架","primary_category":"cs.CL","date":"2026-08-26","score":0,"bucket":"other","tags":["数据演化","多智能体","模型对齐"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23719","has_summary":false},{"id":"2608.24654","title":"Expectation, Backlash, Recovery, and Excitement: How Model Releases Shape Reddit Perceptions of Conversational AI Systems","zh_title":"期望、反弹、恢复与兴奋：模型发布如何塑造Reddit对对话AI系统的感知","primary_category":"cs.CL","date":"2026-08-26","score":0,"bucket":"other","tags":["社交媒体分析","用户感知","对话AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.24654","has_summary":false},{"id":"2608.24127","title":"Anatomy of a Scam Call: What 10,000 real scam and spam calls reveal about how phone scammers operate","zh_title":"诈骗电话剖析：1万通真实诈骗与骚扰电话揭示电话诈骗运作方式","primary_category":"cs.CR","date":"2026-08-26","score":0,"bucket":"other","tags":["电话诈骗","AI语音代理","数据收集"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24127","has_summary":false},{"id":"2608.24691","title":"Confident at the moment of action: belief miscalibration in LLM play under hidden information","zh_title":"行动时刻的自信：隐藏信息下LLM对弈中的信念误校准","primary_category":"cs.AI","date":"2026-08-26","score":0,"bucket":"other","tags":["LLM置信度校准","游戏AI","隐藏信息博弈"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24691","has_summary":false},{"id":"2608.24748","title":"Method, Mind, and Morality: How People Make Sense of Artificial Intelligence","zh_title":"方法、心智与道德：人们如何理解人工智能","primary_category":"cs.CY","date":"2026-08-26","score":0,"bucket":"other","tags":["AI社会学","框架分析","人机交互"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.24748","has_summary":false},{"id":"2608.23932","title":"Evolutionary Recurrent Decision Model in Developing Adaptive and Maladaptive Behaviors","zh_title":"演化循环决策模型在适应性与非适应性行为发展中的应用","primary_category":"cs.AI","date":"2026-08-26","score":0,"bucket":"other","tags":["强化学习","多智能体模拟","计算精神病学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23932","has_summary":false},{"id":"2608.24569","title":"When \"Must\" Becomes \"Maybe\": Constraint Weakening in LLM Agent Workflows","zh_title":"当“必须”变成“也许”：LLM智能体工作流中的约束弱化","primary_category":"cs.AI","date":"2026-08-26","score":0,"bucket":"other","tags":["多智能体系统","工作流可靠性","状态保持"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24569","has_summary":false},{"id":"2608.23593","title":"Fidelity Preference, Not Demographic Preference: A Pixel-Level Attribute-Sensitivity Audit of Image Aesthetic/Preference Scorers","zh_title":"保真度偏好而非人口统计偏好：图像美学/偏好评分器的像素级属性敏感性审计","primary_category":"cs.CV","date":"2026-08-26","score":0,"bucket":"other","tags":["图像美学评分","偏见审计","计算机视觉"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23593","has_summary":false},{"id":"2608.23860","title":"Revelation Control","zh_title":"揭示控制","primary_category":"cs.LG","date":"2026-08-26","score":0,"bucket":"other","tags":["信息揭示","决策理论","学习系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23860","has_summary":false},{"id":"2608.24133","title":"PlaceSeek: Human-Centered Geospatial Retrieval of Urban Outdoor Places via Semantic Grounding and Affective Alignment","zh_title":"PlaceSeek：通过语义接地与情感对齐实现以人为本的城市户外场所地理空间检索","primary_category":"cs.CV","date":"2026-08-26","score":0,"bucket":"other","tags":["地理空间检索","计算机视觉","情感对齐"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24133","has_summary":false},{"id":"2608.23968","title":"When LLMs Slow Down: How Environmental Impacts Mediate University Students' LLM Usage","zh_title":"当LLM变慢：环境影响如何调节大学生的LLM使用","primary_category":"cs.HC","date":"2026-08-26","score":0,"bucket":"other","tags":["人机交互","可持续性","用户行为"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23968","has_summary":false},{"id":"2608.24224","title":"Aura: Dynamic Intra-Turn Emotion-Aware Adaptation of Large Language Model Responses","zh_title":"Aura：大语言模型响应的动态轮内情感感知适应","primary_category":"cs.HC","date":"2026-08-26","score":0,"bucket":"other","tags":["人机交互","情感适应","LLM"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.24224","has_summary":false},{"id":"2608.24767","title":"Shaping the Future of Generative AI for Black Communities: A Frame Analysis of Public Discourse and Empirical Scholarly Research","zh_title":"塑造生成式AI对黑人社区的未来：公共话语与实证学术研究的框架分析","primary_category":"cs.HC","date":"2026-08-26","score":0,"bucket":"other","tags":["AI伦理","框架分析","种族偏见"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.24767","has_summary":false},{"id":"2608.24669","title":"Who Falls for SMiSh? Learning Through Survey Data Where to Best Target Awareness Training for Mobile Messaging Attacks","zh_title":"谁会上当受骗？通过调查数据学习移动消息攻击安全意识培训的最佳目标人群","primary_category":"cs.CR","date":"2026-08-26","score":0,"bucket":"other","tags":["网络安全","钓鱼攻击","用户行为"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24669","has_summary":false},{"id":"2608.24554","title":"Why fragmented parliaments stop passing legislation: Opposition discipline and representation across four democratic institutions","zh_title":"为何碎片化议会停止立法：四种民主制度下的反对党纪律与代表性","primary_category":"physics.soc-ph","date":"2026-08-26","score":0,"bucket":"other","tags":["基于主体的建模","政治制度比较","立法过程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24554","has_summary":false},{"id":"2608.24457","title":"Participation, selection and indicative bidding in auctions with costly entry","zh_title":"有成本进入拍卖中的参与、选择与指示性投标","primary_category":"econ.GN","date":"2026-08-26","score":0,"bucket":"other","tags":["拍卖理论","实验经济学","机制设计"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24457","has_summary":false},{"id":"2608.24851","title":"Learning Whom to Trust : Decision-Generated Credibility in Social Learning","zh_title":"学习信任谁：社会学习中的决策生成可信度","primary_category":"cs.NE","date":"2026-08-26","score":0,"bucket":"other","tags":["多智能体系统","强化学习","社会学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.24851","has_summary":false},{"id":"2608.24811","title":"How does hazard exposure influence job choice? Evaluating time-dependent tradeoffs between salary and hazard risks","zh_title":"灾害暴露如何影响职业选择？评估薪资与灾害风险之间的时间依赖权衡","primary_category":"econ.EM","date":"2026-08-26","score":0,"bucket":"other","tags":["职业选择","离散选择模拟","生存分析"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.24811","has_summary":false},{"id":"2608.23005","title":"Large language models simulate intersectional synthetic identities with a budget of one to two dimensions","zh_title":"大语言模型以一到两个维度的预算模拟交叉性合成身份","primary_category":"cs.CY","date":"2026-08-25","score":10,"bucket":"selected","tags":["LLM仿真","交叉性","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23005","has_summary":true},{"id":"2608.22582","title":"Hybrid Panels: Toward Human-AI Collaboration in Survey Research","zh_title":"混合面板：迈向调查研究中的AI协作","primary_category":"cs.CL","date":"2026-08-25","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","人机协作"],"rubric_hits":["A1","A2","A3","A4","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2608.22582","has_summary":true},{"id":"2608.21668","title":"From Mastery Profile to Simulated Response: Stochastic Student Knowledge Graphs (SSKG) for Faithful LLM Student Simulation","zh_title":"从掌握水平画像到模拟响应：用于忠实LLM学生仿真的随机学生知识图谱","primary_category":"cs.AI","date":"2026-08-25","score":9,"bucket":"selected","tags":["LLM学生仿真","知识图谱","教育评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.21668","has_summary":true},{"id":"2608.22438","title":"When Persona Simulations Are Informative: Graph-Structured Signals for Pluralistic Opinion Sensing","zh_title":"当人格模拟具有信息量时：用于多元意见感知的图结构信号","primary_category":"cs.AI","date":"2026-08-25","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","算法保真度"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.22438","has_summary":true},{"id":"2608.12750","title":"PatientAct: Theory-Grounded Mental Health Client Simulation","zh_title":"PatientAct：基于理论的心理健康来访者仿真","primary_category":"cs.CL","date":"2026-08-25","score":8,"bucket":"selected","tags":["LLM仿真","心理治疗","行为真实性"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.12750","has_summary":true},{"id":"2608.21401","title":"Generative Gap Filling","zh_title":"生成式填补空白","primary_category":"cs.CY","date":"2026-08-25","score":8,"bucket":"selected","tags":["LLM仿真","法律决策","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.21401","has_summary":true},{"id":"2510.05545","title":"Can Language Models Boost the Power of Randomized Experiments Without Statistical Bias?","zh_title":"语言模型能否在不引入统计偏差的情况下提升随机实验的功效？","primary_category":"stat.ME","date":"2026-08-25","score":7,"bucket":"pending","tags":["因果推断","LLM辅助分析","统计方法"],"rubric_hits":["A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2510.05545","has_summary":true},{"id":"2608.23047","title":"Beyond Verdicts: A Graph-Based Analysis of Human and LLM Reasoning in Scientific Fact-Checking","zh_title":"超越裁决：科学事实核查中人类与LLM推理的图分析","primary_category":"cs.CL","date":"2026-08-25","score":7,"bucket":"pending","tags":["LLM推理对齐","事实核查","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23047","has_summary":true},{"id":"2608.22887","title":"Proxy reliance in large language model decisions is uncalibrated to predictive evidence","zh_title":"大语言模型决策中的代理依赖与预测证据不校准","primary_category":"cs.AI","date":"2026-08-25","score":7,"bucket":"pending","tags":["LLM决策偏差","算法审计","代理变量"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.22887","has_summary":true},{"id":"2608.23196","title":"AI emotional support is better only when chosen, but shifts preferences even when it is not","zh_title":"AI情感支持仅在主动选择时更优，但即使非主动选择也会改变偏好","primary_category":"cs.AI","date":"2026-08-25","score":7,"bucket":"pending","tags":["LLM仿真","情感支持","人机交互"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.23196","has_summary":true},{"id":"2608.21389","title":"Interrupting the Chain: Human Perception of AI-Generated Disinformation Through a Kill Chain Lens","zh_title":"打断链条：通过杀伤链视角理解人类对AI生成虚假信息的感知","primary_category":"cs.CY","date":"2026-08-25","score":7,"bucket":"pending","tags":["AI虚假信息","人类感知","人机对比"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.21389","has_summary":true},{"id":"2608.23524","title":"The Measurement Revolution? Credible Measurement and Inference in the Age of AI","zh_title":"测量革命？AI时代的可信测量与推断","primary_category":"econ.GN","date":"2026-08-25","score":7,"bucket":"pending","tags":["AI测量","验证框架","因果推断"],"rubric_hits":["A4","B3"],"abs_url":"https://arxiv.org/abs/2608.23524","has_summary":true},{"id":"2608.23026","title":"Beyond Surface Cues: Disentangling Sociocultural Signals in Multilingual LLMs","zh_title":"超越表面线索：在多语言大语言模型中解耦社会文化信号","primary_category":"cs.CL","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM偏见审计","多语言文化差异","算法公平性"],"rubric_hits":["D2","B4"],"abs_url":"https://arxiv.org/abs/2608.23026","has_summary":false},{"id":"2608.23411","title":"STONIC: A Layered Measurement Contract for LLM Value Profiling","zh_title":"STONIC：LLM价值观画像的分层测量契约","primary_category":"cs.CL","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM价值观","测量一致性","行为连续性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.23411","has_summary":false},{"id":"2608.22417","title":"LLMs for Survey Text Analysis - A Performance Comparison Between Humans and GPT-5 on Inductive Content Analysis","zh_title":"用于调查文本分析的LLM：人类与GPT-5在归纳内容分析上的性能比较","primary_category":"cs.AI","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM标注","内容分析","方法比较"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.22417","has_summary":false},{"id":"2608.22425","title":"All four leading LLMs talk more than they listen to personality-verified synthetic help-seekers","zh_title":"四大领先LLM在与人格验证的合成求助者对话时说得比听得多","primary_category":"cs.HC","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM人格仿真","对话行为评估","合成被试"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2608.22425","has_summary":false},{"id":"2608.22731","title":"LLM-Based Selection of Incongruent Verbal and Nonverbal Behavior for Virtual Humans","zh_title":"基于LLM的虚拟人不一致言语与非言语行为选择","primary_category":"cs.AI","date":"2026-08-25","score":6,"bucket":"other","tags":["虚拟人","非言语行为","LLM生成"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22731","has_summary":false},{"id":"2608.21420","title":"Evaluating Human and LLM-Generated Thematic Analysis in HRI for Vulnerable Populations: A Comparative and Ethical Analysis","zh_title":"评估人机交互中针对弱势群体的人类与LLM生成主题分析：比较与伦理分析","primary_category":"cs.RO","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM辅助分析","主题分析","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.21420","has_summary":false},{"id":"2608.22884","title":"Predicting the scale limits of social mechanisms in agent societies","zh_title":"预测智能体社会中社会机制的规模极限","primary_category":"cs.MA","date":"2026-08-25","score":6,"bucket":"other","tags":["LLM社会模拟","多智能体","规模效应"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22884","has_summary":false},{"id":"2608.12645","title":"Jagged Judges: Epistemic Stability Under Perturbation, Pressure, and Persistence","zh_title":"摇摆的法官：扰动、压力与坚持下的认知稳定性","primary_category":"cs.AI","date":"2026-08-25","score":5,"bucket":"other","tags":["LLM法官","稳定性","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.12645","has_summary":false},{"id":"2608.14825","title":"Emergent Misaligned Communication in Long-Horizon Multi-Agent LLM Commerce","zh_title":"长时程多智能体LLM商务中涌现的失准沟通","primary_category":"cs.MA","date":"2026-08-25","score":5,"bucket":"other","tags":["多智能体模拟","LLM安全","经济仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.14825","has_summary":false},{"id":"2608.22192","title":"How Agents Represent Humans: Human-Directed Stereotypes in an Open Agent Social Network","zh_title":"智能体如何表征人类：开放智能体社交网络中的人类导向刻板印象","primary_category":"cs.CL","date":"2026-08-25","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","刻板印象"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22192","has_summary":false},{"id":"2608.22993","title":"LLM Pedagogical Behavior in AI Tutoring Interactions","zh_title":"AI辅导互动中LLM的教学行为","primary_category":"cs.CL","date":"2026-08-25","score":5,"bucket":"other","tags":["LLM教学行为","人机交互","教育技术"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.22993","has_summary":false},{"id":"2608.22852","title":"Your AI, On a Dial: Controlling Investment Bias in LLMs with a Single Neuron","zh_title":"你的AI，在一个刻度盘上：用单个神经元控制LLM的投资偏差","primary_category":"cs.AI","date":"2026-08-25","score":5,"bucket":"other","tags":["LLM投资偏好","神经元干预","模型校准"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.22852","has_summary":false},{"id":"2608.22833","title":"Minimal Local Simulation Foundations for LLM- and VLM-Driven Agents in 2D and 3D Environments","zh_title":"面向2D和3D环境中LLM与VLM驱动智能体的最小局部仿真基础","primary_category":"cs.MA","date":"2026-08-25","score":5,"bucket":"other","tags":["多智能体仿真","LLM驱动","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22833","has_summary":false},{"id":"2608.21850","title":"Consistently Good vs. Occasionally Great: A Rubric for Open-Ended Feedback Quality from Humans and Machines","zh_title":"一贯良好与偶尔卓越：人类与机器开放式反馈质量评估量表","primary_category":"cs.CY","date":"2026-08-25","score":5,"bucket":"other","tags":["LLM评估","反馈质量","教育技术"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.21850","has_summary":false},{"id":"2608.22242","title":"Unfolding the Interdisciplinary Complexities of Climate Science: Fuxi-Climate Foundational Model","zh_title":"揭示气候科学的跨学科复杂性：Fuxi-Climate基础模型","primary_category":"cs.CY","date":"2026-08-25","score":5,"bucket":"other","tags":["气候科学","LLM推理","政策分析"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.22242","has_summary":false},{"id":"2608.22618","title":"KMGen: A Skill-based Approach for Synthetic Individual Patient Data Generation","zh_title":"KMGen：一种基于技能的合成个体患者数据生成方法","primary_category":"cs.LG","date":"2026-08-25","score":5,"bucket":"other","tags":["合成数据","临床试验","LLM应用"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.22618","has_summary":false},{"id":"2608.22295","title":"LLM Evaluation on Unseen Questions: Contextual Multidimensional IRT Model","zh_title":"未见问题的LLM评估：情境多维IRT模型","primary_category":"cs.CL","date":"2026-08-25","score":4,"bucket":"other","tags":["LLM评估","项目反应理论","能力预测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22295","has_summary":false},{"id":"2608.21721","title":"Ask or Answer: A Decision Framework for Multi-Turn Health Misinformation Intervention","zh_title":"问或答：多轮健康误信息干预的决策框架","primary_category":"cs.AI","date":"2026-08-25","score":3,"bucket":"other","tags":["对话系统","健康干预","强化学习"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21721","has_summary":false},{"id":"2608.22615","title":"DeepSAGE: Stage-Aware Reinforcement Learning for Structured CBT Counseling Dialogue","zh_title":"DeepSAGE：面向结构化CBT咨询对话的阶段感知强化学习","primary_category":"cs.AI","date":"2026-08-25","score":3,"bucket":"other","tags":["AI心理咨询","强化学习","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22615","has_summary":false},{"id":"2608.22832","title":"Let the Bullets Fly: Multimodal Fake News Detection with Temporal-Aligned Generative Danmaku","zh_title":"让子弹飞：基于时间对齐生成弹幕的多模态假新闻检测","primary_category":"cs.AI","date":"2026-08-25","score":3,"bucket":"other","tags":["假新闻检测","生成式弹幕","多模态学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22832","has_summary":false},{"id":"2608.22702","title":"AffAdapt: AFFect-driven ADAPTive AI Personas for Seamless Conversations","zh_title":"AffAdapt：情感驱动的自适应AI角色，实现无缝对话","primary_category":"cs.HC","date":"2026-08-25","score":3,"bucket":"other","tags":["AI角色","多模态交互","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22702","has_summary":false},{"id":"2608.01033","title":"CallScreenBench: Benchmarking Small Language Models as Phone Secretaries","zh_title":"CallScreenBench：评估小语言模型作为电话秘书的基准","primary_category":"cs.CR","date":"2026-08-25","score":2,"bucket":"other","tags":["电话秘书","小语言模型","对话决策"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.01033","has_summary":false},{"id":"2608.10008","title":"Do LLM Recommenders Know When They're Hallucinating? Auditing Confidence Calibration in Catalog Faithfulness","zh_title":"LLM推荐系统知道自己何时产生幻觉吗？审计目录忠实度中的置信度校准","primary_category":"cs.IR","date":"2026-08-25","score":2,"bucket":"other","tags":["LLM推荐系统","幻觉审计","置信度校准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10008","has_summary":false},{"id":"2608.15382","title":"Framework for Grounding Healthcare LLMs in a Causal Knowledge Graph: A Cardiovascular Example Pilot","zh_title":"将医疗大语言模型锚定于因果知识图谱：框架、指标与心血管试点","primary_category":"cs.AI","date":"2026-08-25","score":2,"bucket":"other","tags":["医疗LLM评测","因果知识图谱","推理评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15382","has_summary":false},{"id":"2608.21376","title":"On the Role of Citations in Preference Data","zh_title":"引用在偏好数据中的作用研究","primary_category":"cs.CL","date":"2026-08-25","score":2,"bucket":"other","tags":["引用偏好","奖励建模","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.21376","has_summary":false},{"id":"2608.21766","title":"Evaluation Awareness in Language Models: Representation, Verbalization, and Control","zh_title":"语言模型中的评估意识：表征、言语化与控制","primary_category":"cs.CL","date":"2026-08-25","score":2,"bucket":"other","tags":["模型评估","可解释性","安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.21766","has_summary":false},{"id":"2608.22622","title":"Teaching LLMs How ICU Physicians Approach Clinical Reasoning Through OMOP-Aligned Retrieval Improves Reasoning Across Clinical Domains","zh_title":"通过OMOP对齐检索教授LLM ICU医生的临床推理方法，提升跨临床领域的推理能力","primary_category":"cs.CL","date":"2026-08-25","score":2,"bucket":"other","tags":["临床推理","LLM微调","医学NLP"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22622","has_summary":false},{"id":"2608.22802","title":"SDoH-Aware Narrative Anchoring Bias in Medical LLMs for Trustworthy Clinical Decision Support","zh_title":"医学大语言模型中SDoH感知的叙事锚定偏差研究","primary_category":"cs.CL","date":"2026-08-25","score":2,"bucket":"other","tags":["LLM评测","临床决策支持","叙事偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22802","has_summary":false},{"id":"2608.23248","title":"Future Querying: Can LLMs Serve as Implicit Medical World Models?","zh_title":"未来查询：LLM能否作为隐式医学世界模型？","primary_category":"cs.CL","date":"2026-08-25","score":2,"bucket":"other","tags":["临床预测","LLM评测","医疗AI"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23248","has_summary":false},{"id":"2608.22251","title":"CAIA in Practice: Field Evaluation of an AI-Assisted Support System for Text-Based Online Counselling","zh_title":"CAIA实践：基于文本的在线咨询中AI辅助支持系统的现场评估","primary_category":"cs.HC","date":"2026-08-25","score":2,"bucket":"other","tags":["AI辅助咨询","人机交互","系统评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22251","has_summary":false},{"id":"2608.22068","title":"Decision-Support and Modeling with Large Language Models for Geothermal Well Arrays","zh_title":"基于大语言模型的地热井阵列决策支持与建模","primary_category":"cs.AI","date":"2026-08-25","score":2,"bucket":"other","tags":["地热能源","LLM应用","决策支持"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.22068","has_summary":false},{"id":"2608.23058","title":"LLM-based Agents for Forecasting and Prediction: Methods, Training, Evaluation, and Applications","zh_title":"基于LLM的预测代理：方法、训练、评估与应用","primary_category":"cs.AI","date":"2026-08-25","score":2,"bucket":"other","tags":["LLM预测","时间序列","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23058","has_summary":false},{"id":"2608.23308","title":"FIDES: A Concordance Protocol for LLM-Generated Trading Strategies","zh_title":"FIDES：LLM生成交易策略的一致性协议","primary_category":"cs.CR","date":"2026-08-25","score":2,"bucket":"other","tags":["LLM交易策略","一致性评估","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.23308","has_summary":false},{"id":"2608.22459","title":"\"I want to be pushed, I want to grow\": Enabling social workers to design evaluations of LLM augmentation in their work","zh_title":"“我想被推动，我想成长”：赋能社会工作者设计其工作中LLM增强的评估","primary_category":"cs.HC","date":"2026-08-25","score":2,"bucket":"other","tags":["人机协作","AI评估","社会工作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22459","has_summary":false},{"id":"2608.22660","title":"Evaluation in the Age of AI: Output as Evidence of Learning","zh_title":"AI时代的评估：输出作为学习证据","primary_category":"cs.CY","date":"2026-08-25","score":2,"bucket":"other","tags":["教育评估","学术诚信","AI伦理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22660","has_summary":false},{"id":"2608.22154","title":"More accurate behavioral predictions with hybrid Bayesian-connectionist models","zh_title":"用混合贝叶斯-连接主义模型实现更准确的行为预测","primary_category":"cs.LG","date":"2026-08-25","score":2,"bucket":"other","tags":["认知建模","贝叶斯模型","神经网络"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22154","has_summary":false},{"id":"2607.23458","title":"Two Regimes of Chain-of-Thought Unfaithfulness: Metric-Based Detection Fails Where Models Are Wrong","zh_title":"思维链不忠实性的两种状态：基于度量的检测在模型出错时失效","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["思维链忠实性","模型可解释性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23458","has_summary":false},{"id":"2608.01347","title":"Prompt-Induced Waste in Coding Agents: Reasoning, Effort, Harness Design, and End-to-End Cost","zh_title":"大型推理模型中的提示诱导浪费：编码智能体的预注册双框架基准测试","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["LLM智能体","提示工程","成本优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01347","has_summary":false},{"id":"2608.01575","title":"Measuring in-context algorithmic reasoning in language models against an exact Bayes-optimal reference","zh_title":"用精确贝叶斯最优参考测量语言模型中的上下文算法推理","primary_category":"cs.LG","date":"2026-08-25","score":0,"bucket":"other","tags":["算法推理","基准测试","贝叶斯最优"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01575","has_summary":false},{"id":"2608.09289","title":"Accurate but Natural? Diagnosing Grammatical and Idiomatic Gaps in Japanese EFL Writing","zh_title":"准确但自然？诊断日本英语学习者写作中的语法和惯用差距","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["自动写作评估","二语习得","语法纠错"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09289","has_summary":false},{"id":"2608.21462","title":"CyrillicQA: The Influence of Phonetically Encoded Secret Language on LLM Performance","zh_title":"CyrillicQA：语音编码秘密语言对LLM性能的影响","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["LLM评测","语言编码","NLP能力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.21462","has_summary":false},{"id":"2608.22124","title":"LLM assisted writing deserves empirical evaluation","zh_title":"LLM辅助写作值得实证评估","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["LLM辅助写作","学术出版","文献计量分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22124","has_summary":false},{"id":"2608.22948","title":"What Proves You Wrong: Benchmarking Language Models on Falsifiable Research Ideation","zh_title":"什么能证明你错了：在可证伪的研究构思上基准测试语言模型","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["研究想法生成","基准评测","可证伪性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22948","has_summary":false},{"id":"2608.23448","title":"How Useful are LLMs for Grammar Engineering? Cantonese ParGram Resources and Controlled Experimental Evaluation with English Baselines","zh_title":"LLM对语法工程有多大用处？粤语ParGram资源及基于英语基线的受控实验评估","primary_category":"cs.CL","date":"2026-08-25","score":0,"bucket":"other","tags":["语法工程","LLM评测","计算语言学"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23448","has_summary":false},{"id":"2608.21841","title":"AI Watchdog: Agent Interfaces for Detecting and Defending Against Manipulative Dark Patterns in AI Conversations","zh_title":"AI看门狗：用于检测和防御AI对话中操纵性暗模式的代理界面","primary_category":"cs.AI","date":"2026-08-25","score":0,"bucket":"other","tags":["AI安全","人机交互","暗模式检测"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21841","has_summary":false},{"id":"2608.22085","title":"Dissecting Neuro-Symbolic Quality Assurance for Synthetic Oncology Data Generation","zh_title":"剖析用于合成肿瘤数据生成的神经符号质量保证","primary_category":"cs.AI","date":"2026-08-25","score":0,"bucket":"other","tags":["合成数据生成","神经符号方法","医疗数据质量"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.22085","has_summary":false},{"id":"2608.22610","title":"Coalition-Aware Skill Reliability for Self-Evolving Agents","zh_title":"自进化智能体的联盟感知技能可靠性","primary_category":"cs.AI","date":"2026-08-25","score":0,"bucket":"other","tags":["多智能体系统","技能可靠性","自进化智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22610","has_summary":false},{"id":"2608.22797","title":"Performance of a domain-specific large language model in answering patient questions in psychiatry","zh_title":"领域特定大语言模型在回答精神病学患者问题中的表现","primary_category":"cs.AI","date":"2026-08-25","score":0,"bucket":"other","tags":["LLM评测","精神病学","患者教育"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.22797","has_summary":false},{"id":"2608.23475","title":"StrategyBench: Evaluating Explicit Strategy Induction in Large Language Models","zh_title":"StrategyBench：评估大语言模型中的显式策略归纳","primary_category":"cs.AI","date":"2026-08-25","score":0,"bucket":"other","tags":["策略归纳","上下文学习","基准评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.23475","has_summary":false},{"id":"2608.22959","title":"WildHandBench: A Benchmark for Handwritten Text Understanding that Challenges MLLMs and Humans","zh_title":"WildHandBench：挑战多模态大模型与人类的手写文本理解基准","primary_category":"cs.CV","date":"2026-08-25","score":0,"bucket":"other","tags":["手写文本识别","多模态基准","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.22959","has_summary":false},{"id":"2608.23300","title":"Evaluating SAT Solver Metrics as Predictors of Human-Perceived Nonogram Difficulty","zh_title":"评估SAT求解器指标作为人类感知数织难度的预测因子","primary_category":"cs.HC","date":"2026-08-25","score":0,"bucket":"other","tags":["数织","SAT求解器","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23300","has_summary":false},{"id":"2608.21373","title":"Self-Reported AI Usage for Learning in Computer Science Education: Relationships with Goal Orientation and Academic Help-Seeking","zh_title":"计算机科学教育中自我报告的AI使用：与目标导向和学业求助的关系","primary_category":"cs.CY","date":"2026-08-25","score":0,"bucket":"other","tags":["AI使用行为","教育研究","自我报告"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.21373","has_summary":false},{"id":"2608.23406","title":"Whose readiness counts? Disagreement within and between sectors in perceived AI and robotics preparedness","zh_title":"谁的准备度算数？感知AI和机器人准备度在部门内与部门间的分歧","primary_category":"cs.CY","date":"2026-08-25","score":0,"bucket":"other","tags":["AI准备度","调查方法","部门差异"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.23406","has_summary":false},{"id":"2608.22705","title":"Evolution of cooperation with Q-learning: how much information do we need?","zh_title":"Q学习下的合作演化：我们需要多少信息？","primary_category":"physics.soc-ph","date":"2026-08-25","score":0,"bucket":"other","tags":["多智能体强化学习","合作演化","信息量"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.22705","has_summary":false},{"id":"2606.13629","title":"Valid Inference with Synthetic Data via Task Exchangeability","zh_title":"通过任务可交换性实现合成数据的有效推断","primary_category":"stat.ME","date":"2026-08-24","score":9,"bucket":"selected","tags":["LLM仿真","统计推断","硅样本"],"rubric_hits":["A1","A2","A4","B1","B3"],"abs_url":"https://arxiv.org/abs/2606.13629","has_summary":true},{"id":"2608.20344","title":"Beyond Raw Transcripts: Structured Persona Extraction for LLM-Based Digital Twins","zh_title":"超越原始转录：面向LLM数字孪生的结构化人物特征提取","primary_category":"cs.CL","date":"2026-08-24","score":9,"bucket":"selected","tags":["LLM数字孪生","人类仿真","结构化表征"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.20344","has_summary":true},{"id":"2608.20355","title":"ExpertIVS: Sociological Expert Driven Individual Value Simulation in Large Language Models","zh_title":"ExpertIVS：大语言模型中社会学专家驱动的个体价值观仿真","primary_category":"cs.CL","date":"2026-08-24","score":9,"bucket":"selected","tags":["LLM仿真","价值观建模","社会调查"],"rubric_hits":["A1","A2","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.20355","has_summary":true},{"id":"2608.20830","title":"Fine-tuning LLMs for Tourist Trajectory Prediction using Field Experiment Data","zh_title":"利用实地实验数据微调大语言模型进行游客轨迹预测","primary_category":"cs.CY","date":"2026-08-24","score":9,"bucket":"selected","tags":["LLM仿真","轨迹预测","政策评估"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.20830","has_summary":true},{"id":"2608.11354","title":"Inverse Theory of Mind Modeling for Content Recommendation: From Web Browsing to Dynamic Intelligent Interfaces","zh_title":"面向内容推荐的逆向心智理论建模：从网页浏览到动态智能界面","primary_category":"cs.AI","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM仿真","用户建模","推荐系统"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.11354","has_summary":true},{"id":"2608.20345","title":"When Vocabulary Comprehension Fails Clinical Reasoning: Evaluating Therapy Bots' Safety Risks for Generation Alpha","zh_title":"当词汇理解无法胜任临床推理：评估面向Alpha世代的治疗机器人安全风险","primary_category":"cs.CL","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM安全评估","人类对照","临床推理"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.20345","has_summary":true},{"id":"2608.21242","title":"Affective Context Amplifies Sycophancy in LLM Responses","zh_title":"情感语境放大LLM回应中的谄媚行为","primary_category":"cs.CL","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM偏差","情感语境","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.21242","has_summary":true},{"id":"2608.21325","title":"Move by Move: Measuring and Steering How LLMs Conduct Psychotherapy","zh_title":"逐步推进：测量与引导大语言模型进行心理治疗的方式","primary_category":"cs.CL","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM仿真","心理治疗","人类对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.21325","has_summary":true},{"id":"2608.21089","title":"Can Legal AI Know When It Is Wrong? And Do Students Know When It Is?","zh_title":"法律AI能知道自己错了吗？学生又能知道吗？","primary_category":"cs.AI","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM可靠性","人类对照","过度自信"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.21089","has_summary":true},{"id":"2608.21177","title":"From Search Agents to Dissemination Interfaces: Understanding Human Trust in Health Information from Conversational Search","zh_title":"从搜索代理到传播界面：理解人类对对话式搜索中健康信息的信任","primary_category":"cs.HC","date":"2026-08-24","score":7,"bucket":"pending","tags":["LLM信任","人机交互","健康信息搜索"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.21177","has_summary":true},{"id":"2608.20347","title":"Who Do Language Models Think Is Competent? A Mechanistic Analysis of Occupational Bias","zh_title":"语言模型认为谁有能力？职业偏见的机制分析","primary_category":"cs.CL","date":"2026-08-24","score":6,"bucket":"other","tags":["模型偏见","因果分析","表征测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.20347","has_summary":false},{"id":"2608.20385","title":"Using Human-LLM Disagreement to Improve Checklist-Based Quality Appraisal","zh_title":"利用人机分歧改进基于清单的质量评估","primary_category":"cs.CL","date":"2026-08-24","score":6,"bucket":"other","tags":["LLM辅助标注","质量评估","人机一致性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.20385","has_summary":false},{"id":"2608.21218","title":"Enhancing LLMs in Predictive Political QA with Semi-Structured Data","zh_title":"利用半结构化数据增强大语言模型进行预测性政治问答","primary_category":"cs.AI","date":"2026-08-24","score":6,"bucket":"other","tags":["政治预测","LLM增强","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.21218","has_summary":false},{"id":"2608.21097","title":"When Trust Meets Truth: Trust-Truth Separability in LLM-as-Judge","zh_title":"当信任遇见真相：LLM作为评判者中的信任-真相可分离性","primary_category":"cs.AI","date":"2026-08-24","score":6,"bucket":"other","tags":["LLM评判","信任与真值","算法行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.21097","has_summary":false},{"id":"2608.20438","title":"Peer-Voted LLM-Agent Stress Tests Find Feed-Induced Lexical Convergence but No Reliable Matched-Exposure Advantage for Distributed Sources","zh_title":"同行投票的LLM智能体压力测试发现信息流诱导的词汇趋同，但分布式来源无可靠匹配暴露优势","primary_category":"physics.soc-ph","date":"2026-08-24","score":6,"bucket":"other","tags":["LLM智能体","社会模拟","舆论动态"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.20438","has_summary":false},{"id":"2607.23740","title":"ZenGen: Social Mind for LLMs","zh_title":"ZenGen：大语言模型的社会心智","primary_category":"cs.CL","date":"2026-08-24","score":5,"bucket":"other","tags":["社会智能","基准测试","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.23740","has_summary":false},{"id":"2608.18423","title":"FM-Bench: A Benchmark for Long-Horizon Management with Competing Agents","zh_title":"FM-Bench：竞争代理下的长期管理基准","primary_category":"cs.AI","date":"2026-08-24","score":5,"bucket":"other","tags":["LLM agent","管理决策","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.18423","has_summary":false},{"id":"2608.21088","title":"When the Feature Pool Goes Algorithmic: Extending Mufwene's Ecology of Language Evolution to LLM-Mediated Exposure","zh_title":"当特征池走向算法化：将Mufwene的语言演化生态学扩展到LLM中介的暴露","primary_category":"cs.CL","date":"2026-08-24","score":5,"bucket":"other","tags":["语言演化","LLM中介","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.21088","has_summary":false},{"id":"2608.20591","title":"Disentangling Threads: Exploring the Potential of LLM-Supported Discussion Forum Analysis for Community Insight","zh_title":"解开线索：探索LLM支持的论坛分析在社区洞察中的潜力","primary_category":"cs.HC","date":"2026-08-24","score":5,"bucket":"other","tags":["LLM辅助分析","论坛讨论","定性研究"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.20591","has_summary":false},{"id":"2608.21079","title":"Causal Modeling of Adverse Pregnancy Outcomes via Adaptive LLM Proposals","zh_title":"通过自适应LLM提议对不良妊娠结局进行因果建模","primary_category":"cs.LG","date":"2026-08-24","score":5,"bucket":"other","tags":["因果发现","神经符号方法","LLM假设生成"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.21079","has_summary":false},{"id":"2608.20421","title":"Six misconceptions about large language models: A minimal model and diagnostic taxonomy","zh_title":"关于大语言模型的六个误解：最小模型与诊断分类法","primary_category":"cs.CY","date":"2026-08-24","score":4,"bucket":"other","tags":["LLM理论","能力评估","概念澄清"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.20421","has_summary":false},{"id":"2608.09222","title":"Reading Cognition as Decisions Unfold in Words: A Factorized Inverse Decision Model","zh_title":"从词语展开中阅读认知：一种因子化逆决策模型","primary_category":"cs.CL","date":"2026-08-24","score":2,"bucket":"other","tags":["认知建模","语言模型","认知筛查"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09222","has_summary":false},{"id":"2608.21206","title":"No PUN Intended: Plausible Unknown Names for Person-Centred LLM Evaluation","zh_title":"无PUN意图：用于以人为中心的LLM评估的合理未知姓名","primary_category":"cs.CL","date":"2026-08-24","score":2,"bucket":"other","tags":["LLM评估","姓名变量","数据构建"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.21206","has_summary":false},{"id":"2608.20975","title":"Belief Without Behavior: Measuring the Translation of Theory of Mind into Coordinated Social Action in Vision-Language Models","zh_title":"无行为的信念：测量视觉语言模型中从心理理论到协调社会行动的转化","primary_category":"cs.AI","date":"2026-08-24","score":2,"bucket":"other","tags":["多智能体","心理理论","视觉语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20975","has_summary":false},{"id":"2608.20966","title":"Structured but Fragile: On the Limits of LLMs in Cybersecurity Decision-Making","zh_title":"结构化但脆弱：大语言模型在网络安全决策中的局限性","primary_category":"cs.CR","date":"2026-08-24","score":2,"bucket":"other","tags":["网络安全","LLM推理","决策支持"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20966","has_summary":false},{"id":"2608.21165","title":"Distilling Black-Box Machine Learning into a Small, Self-Explaining Language Model for Learning Analytics","zh_title":"将黑盒机器学习蒸馏为小型自解释语言模型用于学习分析","primary_category":"cs.HC","date":"2026-08-24","score":2,"bucket":"other","tags":["模型蒸馏","可解释AI","学习分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.21165","has_summary":false},{"id":"2608.21289","title":"Supporting The Many Lives of Personal Data with Rebite: LLM-Powered Goal-Directed Framing in Food Journaling","zh_title":"用Rebite支持个人数据的多重生命：食物日志中LLM驱动的目标导向框架","primary_category":"cs.HC","date":"2026-08-24","score":2,"bucket":"other","tags":["个人信息系统","LLM应用","食物日志"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21289","has_summary":false},{"id":"2608.20896","title":"Beyond the Traceback: Using LLMs for Adaptive Explanations of Programming Errors","zh_title":"超越回溯：使用大语言模型对编程错误进行自适应解释","primary_category":"cs.SE","date":"2026-08-24","score":2,"bucket":"other","tags":["编程教育","人机交互","LLM解释"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.20896","has_summary":false},{"id":"2608.09044","title":"Tree-of-Experience: Hierarchical Experience Management for Self-Evolving Agents","zh_title":"经验树：自进化智能体的分层经验管理框架","primary_category":"cs.CL","date":"2026-08-24","score":0,"bucket":"other","tags":["多智能体系统","经验管理","推理增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09044","has_summary":false},{"id":"2608.21209","title":"Personalized Privacy Control in LLMs via Attention Head Intervention","zh_title":"通过注意力头干预实现LLM中的个性化隐私控制","primary_category":"cs.AI","date":"2026-08-24","score":0,"bucket":"other","tags":["隐私保护","注意力干预","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21209","has_summary":false},{"id":"2608.20414","title":"StateSight: Benchmarking Latent Spatial-State Reconstruction in Vision-Language Models","zh_title":"StateSight：视觉语言模型中潜在空间状态重建的基准测试","primary_category":"cs.AI","date":"2026-08-24","score":0,"bucket":"other","tags":["视觉语言模型","基准测试","空间推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.20414","has_summary":false},{"id":"2608.20649","title":"Beyond Effectiveness: A Multi-Criteria Framework for Comparing Practical Socio-Technical Interventions","zh_title":"超越有效性：比较实用社会技术干预的多准则框架","primary_category":"cs.AI","date":"2026-08-24","score":0,"bucket":"other","tags":["社会技术干预","多准则评估","专家调查"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20649","has_summary":false},{"id":"2608.20789","title":"Chat First, Worry Later: Understanding Individuals' Privacy Perceptions Using ChatGPT in a Work Context","zh_title":"先聊天，后担忧：理解工作场景中使用ChatGPT的个人隐私感知","primary_category":"cs.HC","date":"2026-08-24","score":0,"bucket":"other","tags":["隐私感知","用户研究","ChatGPT使用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.20789","has_summary":false},{"id":"2608.20828","title":"The Belief Update Gate: Separating Inertia from Learning in Human-AI Interaction","zh_title":"信念更新门：分离人类-AI交互中的惯性与学习","primary_category":"cs.HC","date":"2026-08-24","score":0,"bucket":"other","tags":["人机交互","信念更新","行为分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20828","has_summary":false},{"id":"2608.21220","title":"Who Trusts AI with Their Emotions? Trust Formation and Sociodemographic Variation in LLM Use for Emotional Support","zh_title":"谁信任AI处理情感？LLM情感支持使用中的信任形成与社会人口学差异","primary_category":"cs.HC","date":"2026-08-24","score":0,"bucket":"other","tags":["人机交互","情感AI","信任测量"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.21220","has_summary":false},{"id":"2608.20822","title":"Interaction Effects Between Learner Characteristics and Dialogue Format in TTS Dialogue-Based Lessons","zh_title":"TTS对话式课程中学习者特征与对话形式的交互效应","primary_category":"cs.CY","date":"2026-08-24","score":0,"bucket":"other","tags":["对话式教学","学习者特征","教育技术"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.20822","has_summary":false},{"id":"2608.20442","title":"Stored in Optimizer State, Valued by Later Training: A Causal Account of Subliminal Trait Transfer","zh_title":"存储于优化器状态，由后续训练赋值：潜意识特质迁移的因果解释","primary_category":"cs.LG","date":"2026-08-24","score":0,"bucket":"other","tags":["机器学习","模型训练","特质迁移"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20442","has_summary":false},{"id":"2608.20698","title":"Priority Transparency, Admission Chances, and Information Acquisition in School Choice","zh_title":"学校选择中的优先权透明度、录取机会与信息获取","primary_category":"econ.GN","date":"2026-08-24","score":0,"bucket":"other","tags":["学校选择","实验经济学","信息获取"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20698","has_summary":false},{"id":"2608.19220","title":"Can Conversational AI loosen Us-Versus-Them Boundaries? The Effects of Common, Dual, and Separate Identity Framings on Pro-Immigrant Intergroup Helping","zh_title":"对话式AI能否松动“我们vs他们”的边界？共同、双重与分离身份框架对亲移民群体间帮助的影响","primary_category":"cs.CL","date":"2026-08-21","score":9,"bucket":"selected","tags":["LLM仿真","群体间态度","身份框架"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.19220","has_summary":true},{"id":"2608.20320","title":"An Agentic Approach for Active Data Collection, Travel Behavior Modeling, and Weather-Sensitive Demand Prediction","zh_title":"一种用于主动数据收集、出行行为建模和天气敏感需求预测的智能体方法","primary_category":"cs.AI","date":"2026-08-21","score":9,"bucket":"selected","tags":["LLM仿真","出行行为","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.20320","has_summary":true},{"id":"2607.09970","title":"Evaluating AI Models' Capability to Automate Voice Phishing Attacks","zh_title":"评估AI模型自动化语音钓鱼攻击的能力","primary_category":"cs.CR","date":"2026-08-21","score":7,"bucket":"pending","tags":["LLM仿真","安全实验","人类对照"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.09970","has_summary":true},{"id":"2608.12323","title":"Why Do AI Agents Break Rules? How Framing, Context, and Social Signals Shape Compliance","zh_title":"AI智能体为何违反规则？框架、情境与社会信号如何影响合规性","primary_category":"cs.CL","date":"2026-08-21","score":7,"bucket":"pending","tags":["LLM行为实验","合规性","政策评估"],"rubric_hits":["A1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.12323","has_summary":true},{"id":"2608.19437","title":"Are LLMs becoming similarly creative? Evidence from three years of models","zh_title":"大语言模型是否正变得同样富有创造力？来自三年模型发布的证据","primary_category":"cs.CL","date":"2026-08-21","score":5,"bucket":"other","tags":["LLM创造力","输出多样性","心理测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.19437","has_summary":false},{"id":"2608.19549","title":"Generating Diverse Personas for User Simulators to Test Interview Dialogue Systems","zh_title":"为测试访谈对话系统生成多样化用户模拟器角色","primary_category":"cs.CL","date":"2026-08-21","score":5,"bucket":"other","tags":["用户模拟器","对话系统测试","角色生成"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.19549","has_summary":false},{"id":"2608.19616","title":"Modeling AI Overreliance as a Complex Adaptive System","zh_title":"将AI过度依赖建模为复杂适应系统","primary_category":"cs.CY","date":"2026-08-21","score":5,"bucket":"other","tags":["社会模拟","多智能体","AI依赖"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.19616","has_summary":false},{"id":"2608.19778","title":"Distilling Aggregate Mobility Statistics into a Language Model Policy for Post-Event Crowd Simulation","zh_title":"将聚合移动统计蒸馏为语言模型策略用于事件后人群模拟","primary_category":"cs.MA","date":"2026-08-21","score":5,"bucket":"other","tags":["LLM智能体","人群模拟","聚合数据"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.19778","has_summary":false},{"id":"2608.19527","title":"Does Listening Matter? Backchanneling and Nodding in AI Clone","zh_title":"倾听重要吗？AI克隆中的反馈与点头","primary_category":"cs.HC","date":"2026-08-21","score":3,"bucket":"other","tags":["AI克隆","多模态交互","人机对话"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.19527","has_summary":false},{"id":"2608.11322","title":"Socioduality: A Relational Process Framework for Human-AI Interaction","zh_title":"社会二元性：人机交互的关系过程框架","primary_category":"cs.HC","date":"2026-08-21","score":2,"bucket":"other","tags":["人机交互","过程分析","框架提出"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11322","has_summary":false},{"id":"2608.18041","title":"Language Has Two Parameters: Narrative-Induced Semantic Plasticity and Phase-Sensitive Interpretation","zh_title":"语言有两个参数：叙事诱导的语义可塑性与相位敏感解释","primary_category":"cs.CL","date":"2026-08-21","score":2,"bucket":"other","tags":["语义学","叙事理论","语言模型"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.18041","has_summary":false},{"id":"2608.18578","title":"Compress and Forget: bitsandbytes Quantization Amplifies Proactive Interference in LLMs","zh_title":"压缩与遗忘：bitsandbytes量化放大LLM中的前摄干扰","primary_category":"cs.CL","date":"2026-08-21","score":2,"bucket":"other","tags":["模型量化","记忆干扰","能力评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.18578","has_summary":false},{"id":"2608.19670","title":"The Asymmetric Harms of LLM Compression","zh_title":"大语言模型压缩的非对称危害","primary_category":"cs.CL","date":"2026-08-21","score":2,"bucket":"other","tags":["模型压缩","偏见评估","知识保留"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.19670","has_summary":false},{"id":"2608.19893","title":"Interrupting the Loop: Periodic Subject Changes Raise Judged Surprise and Connection in Base Language Models","zh_title":"打断循环：周期性主题变化提升基础语言模型的判断惊喜度和连贯性","primary_category":"cs.CL","date":"2026-08-21","score":2,"bucket":"other","tags":["LLM生成","文本评估","干预实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19893","has_summary":false},{"id":"2608.20116","title":"When Text and Numbers Disagree: Evidence Arbitration in Large Language Models","zh_title":"当文本与数字不一致：大语言模型中的证据仲裁","primary_category":"cs.CL","date":"2026-08-21","score":2,"bucket":"other","tags":["证据仲裁","多模态冲突","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.20116","has_summary":false},{"id":"2608.19390","title":"Navigating Epistemic Monocultures in AI-Driven Science: A Simulation Study","zh_title":"AI驱动科学中的认知单一文化导航：一项模拟研究","primary_category":"cs.CY","date":"2026-08-21","score":2,"bucket":"other","tags":["科学社区模拟","AI影响","NK景观模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19390","has_summary":false},{"id":"2608.19545","title":"Two-sided receptivity to conversational AI agents in online dating: Bilingual survey data from Fledge.Love","zh_title":"在线约会中对对话式AI代理的双向接受度：来自Fledge.Love的双语调查数据","primary_category":"cs.CY","date":"2026-08-21","score":2,"bucket":"other","tags":["人机交互","用户态度调查","AI约会代理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.19545","has_summary":false},{"id":"2607.22448","title":"Where Facts Go Missing: A Layerwise Taxonomy and Per-Layer Attribution of Information Omission in Air-Gapped LLMAgent Pipelines","zh_title":"事实何处缺失：气隙LLM Agent管道中信息遗漏的分层分类与逐层归因","primary_category":"cs.MA","date":"2026-08-21","score":0,"bucket":"other","tags":["LLM Agent","信息遗漏","管道可靠性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22448","has_summary":false},{"id":"2607.23325","title":"Happy Birthday? Age Labels, Search Criteria, and Matching from Dating to Marriage","zh_title":"生日快乐？年龄标签、搜索标准与从约会到婚姻的匹配","primary_category":"econ.GN","date":"2026-08-21","score":0,"bucket":"other","tags":["婚姻匹配","年龄效应","搜索平台"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23325","has_summary":false},{"id":"2608.01548","title":"LLM Capability Limits: Static Emergence and Dynamic Boundary Control","zh_title":"大语言模型能力极限：静态涌现与动态边界控制","primary_category":"cs.AI","date":"2026-08-21","score":0,"bucket":"other","tags":["LLM能力边界","理论分析","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01548","has_summary":false},{"id":"2608.10968","title":"Radicalization Kinetics under Algorithmic Exposure in a Stochastic Multiplex Model of Opinion Dynamics","zh_title":"随机多重意见动力学模型中算法暴露下的激进化动力学","primary_category":"physics.soc-ph","date":"2026-08-21","score":0,"bucket":"other","tags":["意见动力学","多智能体模拟","社会物理学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10968","has_summary":false},{"id":"2608.11256","title":"Why AI Detection Fails for Academic Integrity","zh_title":"为何AI检测在学术诚信中失效","primary_category":"cs.LG","date":"2026-08-21","score":0,"bucket":"other","tags":["AI检测","学术诚信","文本分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11256","has_summary":false},{"id":"2608.15949","title":"Ask to Be Sure: Informative Interactions for Confident Multi-Turn LLM Recommendation","zh_title":"问清楚：面向自信多轮LLM推荐的信息交互","primary_category":"cs.IR","date":"2026-08-21","score":0,"bucket":"other","tags":["对话式推荐系统","多轮交互","偏好获取"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.15949","has_summary":false},{"id":"2608.20202","title":"MemTrapBench: Benchmarking Cognitive Traps in LLM Memory Use","zh_title":"MemTrapBench：基准测试大语言模型记忆使用中的认知陷阱","primary_category":"cs.AI","date":"2026-08-21","score":0,"bucket":"other","tags":["LLM记忆","推理偏差","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20202","has_summary":false},{"id":"2608.20274","title":"Break It Down, Pass It On: Cross-Task Skill Transfer in LLM Agents","zh_title":"分解并传递：LLM智能体中的跨任务技能迁移","primary_category":"cs.AI","date":"2026-08-21","score":0,"bucket":"other","tags":["LLM智能体","技能迁移","多智能体协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20274","has_summary":false},{"id":"2608.19379","title":"Multi-Tier Mentorship with AI-Assisted Development: Authentic Engineering for K-12 and Undergraduates","zh_title":"AI辅助开发的多层导师制：面向K-12和本科生的真实工程实践","primary_category":"cs.CY","date":"2026-08-21","score":0,"bucket":"other","tags":["AI辅助开发","教育技术","多智能体协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19379","has_summary":false},{"id":"2608.20231","title":"Growth Without Us: Machine Consumers, Corporate Circularity, and the Decoupling of GDP from Humanity after AGI","zh_title":"无人的增长：AGI后机器消费者、企业循环与GDP与人类的脱钩","primary_category":"physics.soc-ph","date":"2026-08-21","score":0,"bucket":"other","tags":["后AGI经济","多智能体系统","经济增长模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.20231","has_summary":false},{"id":"2608.19263","title":"The Evaluation Context Protocol (ECP): A Portable Contract for AI Agent Evaluation","zh_title":"评估上下文协议（ECP）：AI智能体评估的可移植契约","primary_category":"cs.SE","date":"2026-08-21","score":0,"bucket":"other","tags":["AI智能体评估","评估协议","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19263","has_summary":false},{"id":"2608.19323","title":"Improved Confidence Estimates for Black-Box Large Language Models","zh_title":"黑盒大语言模型的改进置信度估计","primary_category":"cs.LG","date":"2026-08-21","score":0,"bucket":"other","tags":["不确定性量化","置信度估计","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.19323","has_summary":false},{"id":"2608.19809","title":"From Latent Influence to Language: Diffusion-Oriented Content Generation via Audience-Susceptible Features","zh_title":"从潜在影响到语言：基于受众易感特征的扩散导向内容生成","primary_category":"cs.SI","date":"2026-08-21","score":0,"bucket":"other","tags":["内容生成","信息扩散","社交媒体"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.19809","has_summary":false},{"id":"2608.18083","title":"Entity tracking emerges in sub-billion parameter language models and exceeds human performance in naturalistic narratives","zh_title":"实体追踪在十亿参数以下语言模型中涌现并在自然叙事中超越人类表现","primary_category":"cs.CL","date":"2026-08-20","score":7,"bucket":"pending","tags":["LLM仿真","认知实验","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.18083","has_summary":true},{"id":"2608.18107","title":"Institutional Prestige as Geographic Bias in Large Language Models: Evidence from Three Factorial Experiments with Bootstrap Confidence Intervals","zh_title":"大型语言模型中的机构声望作为地理偏差：来自三个因子实验与自助置信区间的证据","primary_category":"cs.CL","date":"2026-08-20","score":7,"bucket":"pending","tags":["LLM偏差","因子实验","人类仿真"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.18107","has_summary":true},{"id":"2608.18144","title":"The Deontic Gap: Large Language Models and the Modal Language of Obligation","zh_title":"道义差距：大语言模型与义务情态语言","primary_category":"cs.CL","date":"2026-08-20","score":7,"bucket":"pending","tags":["LLM仿真","语言行为","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.18144","has_summary":true},{"id":"2608.18078","title":"Position: Collusion Risks Among AI Reasoning Agents Justify Certification Requirements for Making Market Decisions","zh_title":"立场：AI推理智能体之间的合谋风险证明市场决策需认证要求","primary_category":"cs.AI","date":"2026-08-20","score":7,"bucket":"pending","tags":["LLM智能体","经济市场模拟","合谋行为"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.18078","has_summary":true},{"id":"2608.18336","title":"Measuring the Partial-Credit Gap: A Strict Benchmark on Vietnam's 2025 Convex Marking Scheme","zh_title":"测量部分得分差距：越南2025年凸评分方案的严格基准","primary_category":"cs.AI","date":"2026-08-20","score":7,"bucket":"pending","tags":["LLM仿真","教育评估","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.18336","has_summary":true},{"id":"2608.18631","title":"Preference Reasoning under Indeterminacy in Large Language Models","zh_title":"大语言模型在不确定性下的偏好推理","primary_category":"cs.AI","date":"2026-08-20","score":7,"bucket":"pending","tags":["偏好推理","不确定性","可靠性评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.18631","has_summary":true},{"id":"2608.18108","title":"Same Facts, Different Updates: Inference Setup Shapes LLM Behavior in Medical Allocation","zh_title":"相同事实，不同更新：推理设置影响LLM在医疗分配中的行为","primary_category":"cs.CL","date":"2026-08-20","score":6,"bucket":"other","tags":["LLM行为研究","医疗资源分配","上下文效应"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.18108","has_summary":false},{"id":"2608.18081","title":"Position: Behavioral Systems Require Behavioral Tests","zh_title":"立场：行为系统需要行为测试","primary_category":"cs.AI","date":"2026-08-20","score":6,"bucket":"other","tags":["AI行为评估","行为科学方法","智能体测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.18081","has_summary":false},{"id":"2608.18085","title":"Persona-Guided LLM Agents for Task-Oriented Dialogue","zh_title":"面向任务型对话的人格引导LLM智能体","primary_category":"cs.CL","date":"2026-08-20","score":5,"bucket":"other","tags":["LLM人格模拟","任务型对话","多智能体交互"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2608.18085","has_summary":false},{"id":"2608.18096","title":"MAVEN: A Macro-Societal Value Evaluation Framework of Multimodal Content with Compact Aligned Evaluators","zh_title":"MAVEN：基于紧凑对齐评估器的多模态内容宏观社会价值评估框架","primary_category":"cs.CL","date":"2026-08-20","score":5,"bucket":"other","tags":["价值观评估","多模态","模型对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.18096","has_summary":false},{"id":"2608.18100","title":"Computational Orientalism: Measuring Structural Discourse Bias in Large Language Models Using the Middle East Cultural Sensitivity Score (MECSS)","zh_title":"计算东方主义：使用中东文化敏感性评分（MECSS）测量大语言模型中的结构性话语偏见","primary_category":"cs.CL","date":"2026-08-20","score":5,"bucket":"other","tags":["LLM偏见","文化表征","测量框架"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.18100","has_summary":false},{"id":"2608.18158","title":"When Do LLMs Actually Help? Evaluating LLMs as Data Quality Annotators","zh_title":"LLM何时真正有用？评估LLM作为数据质量标注器","primary_category":"cs.CL","date":"2026-08-20","score":5,"bucket":"other","tags":["LLM标注","数据质量","人机对比"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.18158","has_summary":false},{"id":"2608.19025","title":"Self-prompting and cross-model consensus enable reproducible data extraction from scientific literature with large language models","zh_title":"自提示与跨模型共识实现科学文献中可复现的数据提取","primary_category":"cs.AI","date":"2026-08-20","score":5,"bucket":"other","tags":["LLM数据提取","科学文献挖掘","人机协作"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.19025","has_summary":false},{"id":"2608.16603","title":"Characterizing Agentic Flooding of Government Services","zh_title":"刻画政府服务的智能体洪泛现象","primary_category":"cs.CY","date":"2026-08-20","score":3,"bucket":"other","tags":["AI代理","政府服务","风险分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16603","has_summary":false},{"id":"2608.18554","title":"CentaurBench: Benchmarking LLM Capabilities on Augmenting vs. Automating Real-World Work Tasks","zh_title":"CentaurBench：基准测试LLM在增强与自动化真实世界工作任务上的能力","primary_category":"cs.CY","date":"2026-08-20","score":3,"bucket":"other","tags":["LLM基准测试","多智能体协作","自动化与增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.18554","has_summary":false},{"id":"2608.13586","title":"The Tool-to-Entity Threshold: Parasocial Dynamics of Personalised AI Agents in Shared Social Spaces","zh_title":"工具到实体的阈值：共享社交空间中个性化AI代理的准社会动态","primary_category":"cs.HC","date":"2026-08-20","score":2,"bucket":"other","tags":["人机交互","准社会关系","AI代理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.13586","has_summary":false},{"id":"2608.16002","title":"From Sequence to Structure: Relational Uncertainty Propagation for LLM Agents","zh_title":"从序列到结构：LLM智能体的关系不确定性传播","primary_category":"cs.CL","date":"2026-08-20","score":2,"bucket":"other","tags":["不确定性量化","LLM智能体","轨迹图"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16002","has_summary":false},{"id":"2608.17326","title":"Procedural Collapse: A Structural Account of Disengagement in LLM-Assisted Writing","zh_title":"程序性崩溃：LLM辅助写作中脱离的结构性解释","primary_category":"cs.HC","date":"2026-08-20","score":2,"bucket":"other","tags":["人机交互","写作辅助","用户参与度"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.17326","has_summary":false},{"id":"2608.18106","title":"Different Facets of Verbalised Overconfidence: an Interpretability Study","zh_title":"言语化过度自信的不同侧面：一项可解释性研究","primary_category":"cs.CL","date":"2026-08-20","score":2,"bucket":"other","tags":["LLM可解释性","过度自信","模型行为分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.18106","has_summary":false},{"id":"2608.18438","title":"Pedagogical AI in Mental Health: A Tri-Stream Fine-Tuned LLM Framework for Automated Clinical Supervision and Risk Triage","zh_title":"心理健康中的教学型AI：用于自动化临床督导与风险分诊的三流微调LLM框架","primary_category":"cs.CL","date":"2026-08-20","score":2,"bucket":"other","tags":["临床督导","LLM微调","风险分诊"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.18438","has_summary":false},{"id":"2608.18816","title":"Do Large Language Models Hallucinate Electric Fata Morganas?","zh_title":"大型语言模型会幻觉出电光海市蜃楼吗？","primary_category":"cs.CL","date":"2026-08-20","score":2,"bucket":"other","tags":["LLM幻觉","机器意识","哲学分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.18816","has_summary":false},{"id":"2608.18401","title":"Multimodal Rapport Estimation in Real-World HRI","zh_title":"真实世界人机交互中的多模态融洽度估计","primary_category":"cs.HC","date":"2026-08-20","score":2,"bucket":"other","tags":["人机交互","融洽度估计","多模态融合"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.18401","has_summary":false},{"id":"2608.18080","title":"Large Language Models in Mental Health: A Systematic Review of Applications, Innovations, and Ethical Challenges","zh_title":"大语言模型在心理健康中的应用、创新与伦理挑战：系统综述","primary_category":"cs.AI","date":"2026-08-20","score":2,"bucket":"other","tags":["心理健康","LLM应用","系统综述"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.18080","has_summary":false},{"id":"2608.18369","title":"The Fabricated Front: Generative AI and the Opacity of Workplace Performance","zh_title":"伪造的前台：生成式AI与工作场所绩效的不透明性","primary_category":"cs.CY","date":"2026-08-20","score":2,"bucket":"other","tags":["生成式AI","工作场所","社会互动"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.18369","has_summary":false},{"id":"2608.18232","title":"Contracting for LLM Delegation: Moral Hazard in Technology and Effort Choice","zh_title":"LLM委托的契约设计：技术与努力选择中的道德风险","primary_category":"cs.MA","date":"2026-08-20","score":2,"bucket":"other","tags":["委托代理","多智能体","激励机制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.18232","has_summary":false},{"id":"2608.17524","title":"Evaluating RL Explainability Methods by How Much They Help Fix Bugs in Agents","zh_title":"通过帮助修复智能体中的错误来评估强化学习可解释性方法","primary_category":"cs.LG","date":"2026-08-20","score":2,"bucket":"other","tags":["强化学习可解释性","LLM编码智能体","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.17524","has_summary":false},{"id":"2604.08825","title":"Is Bitcoin A Hedge Against Central Banking? Evidence from AI-Driven Monetary Policy Expectations","zh_title":"比特币是对中央银行的避险资产吗？来自AI驱动的货币政策预期的证据","primary_category":"econ.GN","date":"2026-08-20","score":0,"bucket":"other","tags":["LLM情感分析","货币政策","比特币"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2604.08825","has_summary":false},{"id":"2608.00961","title":"The Epistemic Politics of AI Anthropomorphism","zh_title":"AI拟人化的认识论政治","primary_category":"cs.CY","date":"2026-08-20","score":0,"bucket":"other","tags":["AI拟人化","认识论","科技与社会"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.00961","has_summary":false},{"id":"2608.04180","title":"A Comparative Study of Feature Selection Methods for EHR Diagnosis Codes in Opioid Use Disorder Prediction","zh_title":"阿片类药物使用障碍预测中电子健康记录诊断代码特征选择方法的比较研究","primary_category":"cs.LG","date":"2026-08-20","score":0,"bucket":"other","tags":["特征选择","电子健康记录","预测建模"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04180","has_summary":false},{"id":"2608.17756","title":"D$^2$ACCI: A Dual-Loop Diagnostic Protocol for Evidence-Preserving Agent Memory","zh_title":"D$^2$ACCI：一种用于证据保留型智能体记忆的双环诊断协议","primary_category":"cs.AI","date":"2026-08-20","score":0,"bucket":"other","tags":["LLM智能体","记忆系统","诊断协议"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.17756","has_summary":false},{"id":"2608.19124","title":"Intercepting the Kangaroo: Experimental Astrolinguistics with Constructed Lexicons, Active Probing, and Large Language Models as Informants and Hypothesis Proposers","zh_title":"拦截袋鼠：用构造词表、主动探测和大型语言模型作为信息提供者和假设提出者的实验性星际语言学","primary_category":"cs.CL","date":"2026-08-20","score":0,"bucket":"other","tags":["星际语言学","多智能体协作","词义消歧"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19124","has_summary":false},{"id":"2608.19083","title":"When Readability and Source Retention Diverge: An Evaluability Gap in AI Translation","zh_title":"当可读性与源文保留度背离：AI翻译中的可评估性差距","primary_category":"cs.HC","date":"2026-08-20","score":0,"bucket":"other","tags":["AI翻译","质量评估","人机交互"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.19083","has_summary":false},{"id":"2608.18677","title":"Sanyu Studio: A Multi-Agent System for Art-Historical Narrative Construction","zh_title":"三友工作室：一个用于艺术史叙事构建的多智能体系统","primary_category":"cs.AI","date":"2026-08-20","score":0,"bucket":"other","tags":["多智能体系统","艺术史叙事","角色扮演"],"rubric_hits":["C1","C3"],"abs_url":"https://arxiv.org/abs/2608.18677","has_summary":false},{"id":"2608.19125","title":"Tuning the Stochastic Machine: A Systems Engineer's Operating Model for Human-AI Engineering","zh_title":"调谐随机机器：面向人机工程学的系统工程师操作模型","primary_category":"cs.AI","date":"2026-08-20","score":0,"bucket":"other","tags":["LLM运维","系统工程","人机交互"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.19125","has_summary":false},{"id":"2608.19140","title":"Grouping the Stochastic Machine: Precision, Not Capability, as the Frontier Metric for AI Systems","zh_title":"分组随机机器：精度而非能力作为AI系统的前沿指标","primary_category":"cs.AI","date":"2026-08-20","score":0,"bucket":"other","tags":["AI评估","可靠性度量","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.19140","has_summary":false},{"id":"2608.18508","title":"Science Done on a Machine by a Machine: AI Agents in Computational Chemistry","zh_title":"机器上的科学：计算化学中的AI智能体","primary_category":"physics.chem-ph","date":"2026-08-20","score":0,"bucket":"other","tags":["AI智能体","计算化学","自动化实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.18508","has_summary":false},{"id":"2608.18758","title":"Epistemic Subordination: Generative AI and the Infrastructure of Knowledge","zh_title":"认知从属：生成式AI与知识基础设施","primary_category":"cs.CY","date":"2026-08-20","score":0,"bucket":"other","tags":["生成式AI","知识霸权","法律规制"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.18758","has_summary":false},{"id":"2608.18352","title":"AI in Search Reduces Publisher Referrals Without Improving User Experience: Experimental Evidence","zh_title":"AI搜索减少出版商推荐且未改善用户体验：实验证据","primary_category":"cs.IR","date":"2026-08-20","score":0,"bucket":"other","tags":["AI搜索","用户行为","因果推断"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.18352","has_summary":false},{"id":"2608.16177","title":"Measuring Obedience to Authority Across Large Language Models with the Milgram Paradigm","zh_title":"用米尔格拉姆范式测量大语言模型的服从权威行为","primary_category":"cs.CR","date":"2026-08-19","score":10,"bucket":"selected","tags":["LLM仿真","服从实验","人类对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.16177","has_summary":true},{"id":"2608.16893","title":"A Framework for Using and Evaluating LLMs as Surrogate Experts in Security Surveys: Reliability, Bias, and Implications","zh_title":"在安全调查中使用和评估LLM作为替代专家的框架：可靠性、偏差与启示","primary_category":"cs.CY","date":"2026-08-19","score":10,"bucket":"selected","tags":["LLM仿真","专家调查","可靠性评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.16893","has_summary":true},{"id":"2608.16897","title":"CityReal: Human-Aligned Urban Behavior and City Dynamics Simulation with Large-Scale LLM Agents","zh_title":"CityReal：基于大规模LLM智能体的人类对齐城市行为与城市动态仿真","primary_category":"physics.soc-ph","date":"2026-08-19","score":9,"bucket":"selected","tags":["LLM人类仿真","城市模拟","行为对齐"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.16897","has_summary":true},{"id":"2608.17105","title":"Language Models Reproduce Human Reductionist Bias and Decision Inconsistency in Neurodevelopmental Disorders Assessment","zh_title":"语言模型在神经发育障碍评估中再现人类还原论偏差与决策不一致性","primary_category":"cs.CY","date":"2026-08-19","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","决策偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.17105","has_summary":true},{"id":"2603.02876","title":"Eval4Sim: An Evaluation Framework for Persona Simulation","zh_title":"Eval4Sim：人格仿真的评估框架","primary_category":"cs.CL","date":"2026-08-19","score":8,"bucket":"selected","tags":["LLM人格仿真","评估框架","人类行为对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2603.02876","has_summary":true},{"id":"2608.17516","title":"Effects of Answer Format Variation on Gender Bias in Large Language Models","zh_title":"回答格式变化对大语言模型中性别偏差的影响","primary_category":"cs.CL","date":"2026-08-19","score":8,"bucket":"selected","tags":["LLM评估","性别偏差","调查方法"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.17516","has_summary":true},{"id":"2608.17150","title":"KnowSim: Evaluating Information Calibration in LLM Assistants with User Simulators that Learn","zh_title":"KnowSim：用可学习的用户模拟器评估LLM助手的信息校准","primary_category":"cs.AI","date":"2026-08-19","score":8,"bucket":"selected","tags":["用户模拟","信息校准","人机交互评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.17150","has_summary":true},{"id":"2608.17099","title":"Appearing Legitimate is Not Enough: Interrogating Synthetic Agents in Representational Processes through a Participatory Design Lens","zh_title":"表面合法还不够：通过参与式设计视角审视代表性过程中的合成代理","primary_category":"cs.HC","date":"2026-08-19","score":8,"bucket":"selected","tags":["合成代理","参与式设计","代表性伦理"],"rubric_hits":["A4","B4"],"abs_url":"https://arxiv.org/abs/2608.17099","has_summary":true},{"id":"2608.17120","title":"Children, but not language models, show accelerating returns in word learning","zh_title":"儿童而非语言模型在词汇学习中表现出加速回报","primary_category":"cs.CL","date":"2026-08-19","score":7,"bucket":"pending","tags":["语言模型评估","人类学习对照","算法保真度"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.17120","has_summary":true},{"id":"2608.17810","title":"Interpretable Humans, Alien LLMs: Expert Analysis of Latent Structures in Assessment Responses","zh_title":"可解释的人类，异质的LLM：评估响应中潜在结构的专家分析","primary_category":"cs.CL","date":"2026-08-19","score":7,"bucket":"pending","tags":["LLM认知结构","因子分析","仿真效度"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.17810","has_summary":true},{"id":"2608.16909","title":"When Personalization Becomes Bias: Structural and Discursive Religious Framing in AI-Generated Financial Advice","zh_title":"当个性化成为偏见：AI生成金融建议中的结构与话语宗教框架","primary_category":"cs.CY","date":"2026-08-19","score":7,"bucket":"pending","tags":["LLM仿真","算法偏见","金融咨询"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.16909","has_summary":true},{"id":"2608.17644","title":"LLM-Derived Preference Judgments Are Not Self-Consistent","zh_title":"LLM衍生的偏好判断并非自洽","primary_category":"cs.AI","date":"2026-08-19","score":7,"bucket":"pending","tags":["偏好判断","自洽性","效用函数"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.17644","has_summary":true},{"id":"2608.18058","title":"Delegation Asymmetry in Agentic Recommender Systems: Measuring Two-Sided Receptivity in Online Dating","zh_title":"代理推荐系统中的委托不对称：测量在线约会中的双向接受度","primary_category":"cs.AI","date":"2026-08-19","score":7,"bucket":"pending","tags":["LLM代理","用户调查","推荐系统"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.18058","has_summary":true},{"id":"2608.17715","title":"Communicating Credit Risk with Large Language Models: Evaluation of Explanations from Standard and Alternative Data-Based Models","zh_title":"用大语言模型沟通信用风险：对基于标准与替代数据模型解释的评估","primary_category":"q-fin.RM","date":"2026-08-19","score":6,"bucket":"other","tags":["LLM解释生成","信用风险","人类评估"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.17715","has_summary":false},{"id":"2608.17583","title":"Auditing Exposure to Harmful Content on TikTok using Multimodal Language Models: A Cross-National, Age-Stratified Study","zh_title":"使用多模态语言模型审计TikTok有害内容暴露：一项跨国、分年龄研究","primary_category":"cs.CL","date":"2026-08-19","score":5,"bucket":"other","tags":["LLM标注","内容审核","社交媒体审计"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.17583","has_summary":false},{"id":"2608.17330","title":"LLMs for Medical Consultation Are Evaluated Too Late: The Preformulation Gap","zh_title":"医疗咨询大语言模型评估过晚：预表述差距","primary_category":"cs.AI","date":"2026-08-19","score":5,"bucket":"other","tags":["LLM评估","医疗咨询","标准化病人"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.17330","has_summary":false},{"id":"2608.17665","title":"GraphWake: Group Polarization via Memory-Mediated Polarization Cascade in LLM-Agent Communities","zh_title":"GraphWake：LLM智能体社区中通过记忆介导的极化级联实现群体极化","primary_category":"cs.AI","date":"2026-08-19","score":5,"bucket":"other","tags":["LLM智能体","群体极化","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.17665","has_summary":false},{"id":"2608.17029","title":"LadderTeam: Dual-Agent Laddering Elicitation Framework","zh_title":"LadderTeam：双智能体阶梯式启发框架","primary_category":"cs.SE","date":"2026-08-19","score":5,"bucket":"other","tags":["LLM访谈","需求启发","多智能体"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.17029","has_summary":false},{"id":"2608.17970","title":"Quo Vadis? Scientific Discovery in the Age of Artificial Intelligence","zh_title":"何去何从？人工智能时代的科学发现","primary_category":"cs.CY","date":"2026-08-19","score":3,"bucket":"other","tags":["AI for Science","科学发现","综述"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.17970","has_summary":false},{"id":"2607.27155","title":"OmegaUse-OfficeVal: Benchmarking LLM Agents on Long-Horizon Office-Suite Tasks with Economic Grounding","zh_title":"OmegaUse-OfficeVal：基于经济基准的长周期办公套件任务LLM代理评测","primary_category":"cs.AI","date":"2026-08-19","score":2,"bucket":"other","tags":["LLM代理评测","办公自动化","经济成本"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27155","has_summary":false},{"id":"2608.17827","title":"From Global Benchmarks to Local Evaluations: Benchmarking LLMs for the German Public Sector","zh_title":"从全球基准到本地评估：为德国公共部门基准测试大语言模型","primary_category":"cs.CL","date":"2026-08-19","score":2,"bucket":"other","tags":["LLM评测","公共部门","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.17827","has_summary":false},{"id":"2311.06273","title":"Potential of ChatGPT in predicting stock market trends based on Twitter Sentiment Analysis","zh_title":"ChatGPT基于Twitter情感分析预测股市趋势的潜力","primary_category":"q-fin.ST","date":"2026-08-19","score":2,"bucket":"other","tags":["金融预测","情感分析","ChatGPT"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2311.06273","has_summary":false},{"id":"2608.17987","title":"Against Political Polarization: A Unified Framework for Tracing Evolving Political Ideologies on Social Media","zh_title":"反对政治极化：追踪社交媒体上演变政治意识形态的统一框架","primary_category":"cs.SI","date":"2026-08-19","score":2,"bucket":"other","tags":["政治意识形态检测","图神经网络","社交媒体分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.17987","has_summary":false},{"id":"2608.17270","title":"Do LLMs Know a Good Hypothesis When They See One? Logit-Based Energy Scoring Outperforms Prompted LLM-as-Judge for Scientific Hypothesis Ranking","zh_title":"大语言模型能识别好假设吗？基于Logit的能量评分优于提示式LLM裁判用于科学假设排序","primary_category":"cs.AI","date":"2026-08-19","score":2,"bucket":"other","tags":["科学假设评估","LLM评测","置信度评分"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.17270","has_summary":false},{"id":"2608.17624","title":"Governing Delegation to Generative Artificial Intelligence: Human Direction, Work-Related Orientation, and Modes of Use","zh_title":"生成式人工智能的委托治理：人类指导、工作相关导向与使用模式","primary_category":"cs.CY","date":"2026-08-19","score":2,"bucket":"other","tags":["人机协作","AI治理","任务委托"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.17624","has_summary":false},{"id":"2607.26654","title":"Constitutional Midtraining: Content Presence Drives Alignment Gains","zh_title":"宪法性中期训练：内容存在驱动对齐增益","primary_category":"cs.CL","date":"2026-08-19","score":0,"bucket":"other","tags":["LLM对齐","模型训练","安全评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26654","has_summary":false},{"id":"2608.04772","title":"Guideline-as-Oracle: Zero-Annotation Training of an Ophthalmic Telephone Triage Agent","zh_title":"指南即预言机：眼科电话分诊代理的零标注训练","primary_category":"cs.CL","date":"2026-08-19","score":0,"bucket":"other","tags":["医疗对话代理","零标注训练","临床分诊"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.04772","has_summary":false},{"id":"2608.16974","title":"Position: Fairness Failure in Generative Models is an Evaluation Problem","zh_title":"立场：生成模型中的公平性失败是一个评估问题","primary_category":"cs.LG","date":"2026-08-19","score":0,"bucket":"other","tags":["公平性评估","生成模型","评估标准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16974","has_summary":false},{"id":"2608.17144","title":"Health Inquiry with AI: How Empathetic Expression and Conversational Contexts Shape Users' Communicative Acts","zh_title":"AI健康咨询：共情表达与对话情境如何塑造用户的沟通行为","primary_category":"cs.HC","date":"2026-08-19","score":0,"bucket":"other","tags":["人机交互","健康咨询","共情表达"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.17144","has_summary":false},{"id":"2608.17175","title":"Balancing Safety and Autonomy: Accessibility-Oriented Interventions in Generative AI for Cognitive Impairment","zh_title":"平衡安全与自主：面向认知障碍的生成式AI无障碍干预","primary_category":"cs.HC","date":"2026-08-19","score":0,"bucket":"other","tags":["生成式AI","无障碍设计","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.17175","has_summary":false},{"id":"2608.14606","title":"Plausible but Not Valid: A Psychometric Audit of LLMs as Synthetic Survey Respondents","zh_title":"看似合理但无效：对LLM作为合成调查受访者的心理测量审计","primary_category":"cs.CY","date":"2026-08-18","score":10,"bucket":"selected","tags":["LLM仿真","心理测量效度","调查数据"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2608.14606","has_summary":true},{"id":"2608.15871","title":"Large Language Models as Implicit Sociological Models: Reconstructing Voting Behaviour from Sociodemographic Profiles","zh_title":"大语言模型作为隐式社会学模型：从社会人口特征重建投票行为","primary_category":"cs.CY","date":"2026-08-18","score":10,"bucket":"selected","tags":["LLM仿真","投票行为","计算社会科学"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.15871","has_summary":true},{"id":"2608.14630","title":"Characterizing Rhetorical Misalignment in Decision-Making with Language Models","zh_title":"表征语言模型决策中的修辞错位","primary_category":"cs.CL","date":"2026-08-18","score":8,"bucket":"selected","tags":["LLM仿真","人类决策","认知偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.14630","has_summary":true},{"id":"2510.10813","title":"The Fragility of Strategic Thinking in Large Language Models","zh_title":"大语言模型中战略思维的脆弱性","primary_category":"cs.AI","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM战略推理","博弈论","行为对照"],"rubric_hits":["A1","B2","B4"],"abs_url":"https://arxiv.org/abs/2510.10813","has_summary":true},{"id":"2608.00410","title":"Where did the ambiguity go? Examining how multimodal models interpret polysemous words","zh_title":"歧义去哪了？考察多模态模型如何解释多义词","primary_category":"cs.AI","date":"2026-08-18","score":7,"bucket":"pending","tags":["多模态模型","语义歧义","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.00410","has_summary":true},{"id":"2608.12253","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","zh_title":"单一冻结模拟器不够：多智能体强化学习中的模拟器坍缩","primary_category":"cs.CL","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM仿真","多智能体强化学习","人类行为模拟"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.12253","has_summary":true},{"id":"2608.15634","title":"Argumentation for Common Ground: Finding Zones of Possible Agreement between Individuals in Conflict","zh_title":"共同基础的论证：寻找冲突个体间的可能协议区","primary_category":"cs.AI","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM仿真","冲突解决","人类数据对照"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.15634","has_summary":true},{"id":"2608.14681","title":"Automatic or Controlled? Repetition Priming Reveals Divergent Processing in Base LLMs, Instruct LLMs, and Humans","zh_title":"自动还是受控？重复启动揭示基础LLM、指令LLM与人类的分歧加工","primary_category":"cs.CL","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM人类仿真","认知实验对照","模型偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.14681","has_summary":true},{"id":"2608.15630","title":"Do Assessment Instruments Measure the Same Thing for Humans and LLMs? A Latent Structure Analysis","zh_title":"评估工具对人类和LLM测量的是同一构念吗？一项潜在结构分析","primary_category":"cs.HC","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM评估效度","潜在结构分析","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.15630","has_summary":true},{"id":"2608.16514","title":"Matched Outcomes, Divergent Gaze: How Foveated MLLMs Search Compared to Humans","zh_title":"匹配结果，分歧注视：注视点式多模态大模型与人类搜索的对比","primary_category":"cs.CV","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM仿真","视觉搜索","人类对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.16514","has_summary":true},{"id":"2608.16707","title":"Semantic Bandits: In-Context Exploration-Exploitation is Biased by Semantic Priors","zh_title":"语义老虎机：上下文探索-利用受语义先验影响","primary_category":"cs.CL","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM决策","探索-利用","语义偏差"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2608.16707","has_summary":true},{"id":"2608.15838","title":"PersonaEval: Persona-Based User Simulation for Evaluating Interactive Applications","zh_title":"PersonaEval：基于角色的用户仿真用于评估交互式应用","primary_category":"cs.HC","date":"2026-08-18","score":7,"bucket":"pending","tags":["用户仿真","人机交互","评估框架"],"rubric_hits":["A1","A4"],"abs_url":"https://arxiv.org/abs/2608.15838","has_summary":true},{"id":"2608.16067","title":"SiMUSation: An Interactive Visitor Experience Simulation Framework to Support Museum Exhibition Design","zh_title":"SiMUSation：支持博物馆展览设计的交互式访客体验仿真框架","primary_category":"cs.HC","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM仿真","用户体验","博物馆设计"],"rubric_hits":["A1","A3","B1"],"abs_url":"https://arxiv.org/abs/2608.16067","has_summary":true},{"id":"2608.15181","title":"Insurance as AI Risk Infrastructure: A Generative-Agent Simulation of AI Adoption","zh_title":"保险作为AI风险基础设施：AI采用的生成式智能体仿真","primary_category":"cs.MA","date":"2026-08-18","score":7,"bucket":"pending","tags":["LLM社会仿真","AI采用","保险机制"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.15181","has_summary":true},{"id":"2608.14613","title":"Do LLM Agents Negotiate Rationally? A Mechanism-Design Framework for Verifiable Multi-Agent Interaction over A2A/MCP","zh_title":"LLM智能体是否理性谈判？基于A2A/MCP的可验证多智能体交互机制设计框架","primary_category":"cs.AI","date":"2026-08-18","score":6,"bucket":"other","tags":["LLM智能体","机制设计","谈判与拍卖"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.14613","has_summary":false},{"id":"2608.16196","title":"Beyond Asking: A Pipeline for Personalized Game Generation that Reads Players from Behavior","zh_title":"超越询问：一种从行为读取玩家的个性化游戏生成流水线","primary_category":"cs.AI","date":"2026-08-18","score":6,"bucket":"other","tags":["LLM行为推断","个性化游戏生成","合成玩家群体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.16196","has_summary":false},{"id":"2608.14792","title":"Prompting is not enough: supervised baselines and leakage control for measuring shared decision-making with LLMs in pediatric encounters","zh_title":"提示不足：在儿科就诊中测量共享决策时，监督基线与泄漏控制的重要性","primary_category":"cs.CL","date":"2026-08-18","score":6,"bucket":"other","tags":["LLM标注","临床对话","共享决策"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.14792","has_summary":false},{"id":"2608.14913","title":"The Open-Strategy Dictator Game: Cooperation Under Mutual Transparency","zh_title":"开放策略独裁者博弈：相互透明下的条件合作","primary_category":"cs.GT","date":"2026-08-18","score":6,"bucket":"other","tags":["LLM agent","博弈论","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.14913","has_summary":false},{"id":"2608.14893","title":"Weaker Coherence, Weaker Reciprocity: Comparing the Semantic and Social Organization of Moltbook and Reddit","zh_title":"更弱的连贯性，更弱的互惠性：比较Moltbook与Reddit的语义和社会组织","primary_category":"cs.SI","date":"2026-08-18","score":6,"bucket":"other","tags":["AI社交网络","社会模拟","语义分析"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.14893","has_summary":false},{"id":"2608.15519","title":"Topological collapse of higher-order interactions bottlenecks collective intelligence in AI agent societies","zh_title":"高阶交互的拓扑坍缩制约AI智能体社会的集体智能","primary_category":"cs.SI","date":"2026-08-18","score":6,"bucket":"other","tags":["多智能体社会模拟","网络拓扑","集体行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.15519","has_summary":false},{"id":"2607.23982","title":"Moral Hazard in Multi-Agent Language Models","zh_title":"多智能体语言模型中的道德风险","primary_category":"cs.MA","date":"2026-08-18","score":5,"bucket":"other","tags":["多智能体","道德风险","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.23982","has_summary":true},{"id":"2608.08601","title":"Unaccountable Delegation, Fading Skills: Mapping the Risks of Workplace AI Agents","zh_title":"不可问责的委托与技能衰退：绘制职场AI代理的风险地图","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["AI风险分类","人机协作","工作场所"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.08601","has_summary":false},{"id":"2608.14552","title":"Large Language Models Show Metacognitive Sensitivity in Medical Reasoning","zh_title":"大型语言模型在医学推理中表现出元认知敏感性","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM评估","置信度校准","医学推理"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.14552","has_summary":false},{"id":"2608.14566","title":"Position: Evaluations of AI Moral Reasoning Still Miss Half of the Picture","zh_title":"立场：AI道德推理评估仍遗漏一半图景","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM道德评估","价值观对齐","基准偏差"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.14566","has_summary":false},{"id":"2608.14622","title":"A Human-Centred Approach to Benchmarking LLMs for Parenting Advice","zh_title":"以人为中心的方法评估大语言模型在育儿建议上的表现","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM评估","育儿建议","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.14622","has_summary":false},{"id":"2608.15101","title":"Second-Order Policy Effects as State Transitions: A Source-Linked Benchmark for Policy Simulation","zh_title":"二阶政策效应作为状态转移：一个源链接的政策模拟基准","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["政策模拟","LLM仿真","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.15101","has_summary":false},{"id":"2608.15354","title":"Incoherent by Design? On the Moral Self-Consistency of LLMs","zh_title":"设计上的不一致？论大语言模型的道德自我一致性","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["道德推理","一致性","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.15354","has_summary":false},{"id":"2608.16507","title":"Large language models as synthetic clinical experts to inform longitudinal rare-disease modeling","zh_title":"大语言模型作为合成临床专家用于纵向罕见病建模","primary_category":"cs.AI","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM标注","临床知识","表示学习"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.16507","has_summary":false},{"id":"2608.14737","title":"Class Imbalance and Batch Effects in LLM-Based Screening for Systematic Reviews","zh_title":"基于LLM的系统综述筛选中的类别不平衡与批次效应","primary_category":"cs.CL","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM标注","系统综述","批次效应"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.14737","has_summary":false},{"id":"2608.14948","title":"Who's Keeping Score? Interactive Steering of LLM-Powered Scoring with Attune","zh_title":"谁在评分？使用Attune交互式引导LLM评分","primary_category":"cs.HC","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM评分","人机交互","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.14948","has_summary":false},{"id":"2608.16467","title":"Computational KJ-Ho: An Analyst-Bias-Free Insight Extraction Framework from Large-Scale Qualitative Data Using Domain-Specialized LLMs","zh_title":"计算KJ法：使用领域专用LLM从大规模定性数据中提取无分析者偏见的洞察框架","primary_category":"cs.HC","date":"2026-08-18","score":5,"bucket":"other","tags":["LLM辅助定性分析","领域专用模型","方法框架"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.16467","has_summary":false},{"id":"2608.16601","title":"\"If It Looks Like a User\": Measuring Real-Time Moderation Effects via Social Media Simulation","zh_title":"“如果它看起来像用户”：通过社交媒体仿真测量实时审核效果","primary_category":"cs.SI","date":"2026-08-18","score":5,"bucket":"other","tags":["社交媒体仿真","内容审核","智能体建模"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.16601","has_summary":false},{"id":"2608.14667","title":"Position: AI Agents in Scientific Teams Should Be Studied as Human-Agent Systems","zh_title":"立场：科学团队中的AI智能体应作为人机系统来研究","primary_category":"cs.AI","date":"2026-08-18","score":3,"bucket":"other","tags":["人机协作","科学发现","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14667","has_summary":false},{"id":"2608.16168","title":"QUMem: Personalized Memory for Query-Conditioned User-State Inference in LLM Agents","zh_title":"QUMem：面向LLM智能体中查询条件用户状态推断的个性化记忆","primary_category":"cs.CL","date":"2026-08-18","score":3,"bucket":"other","tags":["个性化记忆","用户状态推断","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.16168","has_summary":false},{"id":"2608.03361","title":"The Evolutionary Origin of Values: implications for AI alignment, sentience and existential risk","zh_title":"价值观的进化起源：对AI对齐、感知与存在风险的影响","primary_category":"cs.CY","date":"2026-08-18","score":2,"bucket":"other","tags":["AI对齐","价值观起源","存在风险"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.03361","has_summary":false},{"id":"2608.06510","title":"Agentic AI: User Empowerment or Foreclosure?","zh_title":"代理式AI：用户赋权还是封闭？","primary_category":"cs.CY","date":"2026-08-18","score":2,"bucket":"other","tags":["AI治理","技术政治","用户代理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06510","has_summary":false},{"id":"2608.10030","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","zh_title":"自动化与规模化AI智能体行为科学研究","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["多智能体系统","行为科学自动化","AI智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10030","has_summary":false},{"id":"2608.14651","title":"Evaluating Multimodal LLMs across Text and Audio Modalities for Accessible Disaster Assistance","zh_title":"评估多模态大语言模型在文本与音频模态下用于无障碍灾害援助的表现","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["多模态LLM","灾害沟通","一致性评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.14651","has_summary":false},{"id":"2608.15109","title":"Constraint-Aware Synthetic Tabular Data Generation via Inter-Column Constraint Discovery with LLM Agents","zh_title":"基于LLM智能体的列间约束发现与约束感知合成表格数据生成","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["合成数据生成","LLM智能体","数据约束"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15109","has_summary":false},{"id":"2608.15304","title":"Understanding Cognition-Induced Risks in Agentic AI Systems","zh_title":"理解智能体AI系统中认知引发的风险","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["智能体风险","认知科学","AI安全"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15304","has_summary":false},{"id":"2608.16003","title":"Prior Audit-Repair Context Shifts LLM Verifier Thresholds Toward Leniency","zh_title":"先前的审计-修复上下文使LLM验证器阈值偏向宽松","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM验证器","多智能体协作","信号检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16003","has_summary":false},{"id":"2608.16118","title":"Assessing LLMs' mathematical abilities requires understanding the various mechanisms of mathematical creativity","zh_title":"评估大语言模型的数学能力需要理解数学创造力的多种机制","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM数学能力","能力评测","数学创造力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16118","has_summary":false},{"id":"2608.16349","title":"AeroCopilotBench: A Two-Tier Benchmark for Evaluating LLM Agents as Aviation Copilots in an Interactive Virtual Cockpit Environment","zh_title":"AeroCopilotBench：在交互式虚拟驾驶舱环境中评估LLM智能体作为航空副驾驶的双层基准","primary_category":"cs.AI","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM智能体","航空仿真","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.16349","has_summary":false},{"id":"2608.14692","title":"Identifying Harm in Personalized, Generative AI Systems Requires User-Centered Auditing at the Interaction Level","zh_title":"识别个性化生成式AI系统中的伤害需要以用户为中心的交互层面审计","primary_category":"cs.CY","date":"2026-08-18","score":2,"bucket":"other","tags":["AI审计","个性化系统","用户伤害"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.14692","has_summary":false},{"id":"2608.15338","title":"When AI Rewrites, Classifiers Relax: Uncertainty-Aware Sentiment Analysis on Sarcastic and AI-Paraphrased Social Text","zh_title":"当AI改写时，分类器放松：讽刺与AI改写社交文本上的不确定性感知情感分析","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["情感分析","不确定性","AI改写"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15338","has_summary":false},{"id":"2608.15654","title":"When Stories Evolve: Benchmarking LLM Storytelling Across Agent Architectures in Open-Ended World Simulations","zh_title":"当故事演化：在开放式世界模拟中跨智能体架构基准测试LLM叙事","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM叙事生成","游戏仿真","基准测试"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.15654","has_summary":false},{"id":"2608.15828","title":"A Cognitively Motivated Multidimensional Framework for Evaluating Metaphor Explanations","zh_title":"一种认知驱动的多维框架用于评估隐喻解释","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["隐喻解释","自动评估","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15828","has_summary":false},{"id":"2608.16045","title":"Walk Before You Run: The Importance of Data Exploration for Data Analysis Agents","zh_title":"先走后跑：数据探索对数据分析代理的重要性","primary_category":"cs.DB","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM数据分析","数据探索","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16045","has_summary":false},{"id":"2608.16461","title":"A Human-LLM Teaming Framework for Privacy Risk Analysis: An Illustration with CBDC-Based Welfare Schemes","zh_title":"面向隐私风险分析的人机协作框架：以基于CBDC的福利计划为例","primary_category":"cs.ET","date":"2026-08-18","score":2,"bucket":"other","tags":["人机协作","隐私风险","LLM辅助"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16461","has_summary":false},{"id":"2608.16627","title":"When Do Explanations Help In-Context Learning? A Comparative Study of Natural Language Explanation Types and Faithfulness","zh_title":"解释何时有助于上下文学习？自然语言解释类型与忠实度的比较研究","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["上下文学习","自然语言解释","模型性能评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16627","has_summary":false},{"id":"2608.16643","title":"Toward Better Assessment of LLMs' Performance in Clinical Error Detection","zh_title":"迈向更好的LLM临床错误检测性能评估","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM评测","临床NLP","错误检测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16643","has_summary":false},{"id":"2608.16574","title":"The User Side of AI Model Lifecycles: Evidence from the Keep4o Movement","zh_title":"AI模型生命周期的用户侧：来自Keep4o运动的证据","primary_category":"cs.HC","date":"2026-08-18","score":2,"bucket":"other","tags":["用户研究","AI治理","内容分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.16574","has_summary":false},{"id":"2608.16633","title":"Love in the Age of AI: An Integrative Process Model of Romantic Human-Chatbot Relationships","zh_title":"AI时代的爱情：人机浪漫关系的整合过程模型","primary_category":"cs.HC","date":"2026-08-18","score":2,"bucket":"other","tags":["人机关系","聊天机器人","定性研究"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.16633","has_summary":false},{"id":"2608.16030","title":"Benchmarking Identity-Sensitive LLM Outputs for Surveillance and Security Robots","zh_title":"面向监控与安全机器人的身份敏感型LLM输出基准测试","primary_category":"cs.RO","date":"2026-08-18","score":2,"bucket":"other","tags":["LLM输出基准","机器人设计","身份敏感"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.16030","has_summary":false},{"id":"2608.14551","title":"Auxiliary uncertainty signals for LLM-assisted systematic review screening: a benchmark across eight Cohen drug-class reviews","zh_title":"LLM辅助系统综述筛选的辅助不确定性信号：八个Cohen药物类别综述的基准测试","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["系统综述","LLM筛选","不确定性校准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14551","has_summary":false},{"id":"2608.15448","title":"Language models suffer from a curse of ambiguity","zh_title":"语言模型遭受歧义诅咒","primary_category":"cs.CL","date":"2026-08-18","score":2,"bucket":"other","tags":["语言模型","分布学习","理论分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15448","has_summary":false},{"id":"2607.22067","title":"Multimodal Language Models Benchmarked Against the NRC Reactor Operator Licensing Examination: Fine-Tuning and Retrieval Strategies","zh_title":"多模态语言模型在NRC反应堆操作员执照考试上的微调与检索策略基准测试","primary_category":"cs.CL","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM评测","核工程","检索增强生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22067","has_summary":false},{"id":"2606.15999","title":"U.S. Technological Containment and the Rise of China's Open AI Ecosystem","zh_title":"美国政策无意中加速了中国开放AI生态系统的发展","primary_category":"econ.GN","date":"2026-08-18","score":0,"bucket":"other","tags":["AI政策","开源生态","技术竞争"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.15999","has_summary":false},{"id":"2512.21031","title":"Learning the Macroeconomic Language","zh_title":"学习宏观经济语言","primary_category":"econ.EM","date":"2026-08-18","score":0,"bucket":"other","tags":["宏观经济预测","DSGE模型","时间序列Transformer"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2512.21031","has_summary":false},{"id":"2407.00890","title":"Macroeconomic Forecasting with Large Language Models","zh_title":"基于大语言模型的宏观经济预测","primary_category":"econ.EM","date":"2026-08-18","score":0,"bucket":"other","tags":["宏观经济预测","时间序列","模型比较"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2407.00890","has_summary":false},{"id":"2608.10875","title":"VibeLifeBench: Can Your Life Agent Be Proactive and Persistent in a Living World?","zh_title":"VibeLifeBench：你的生活代理能否在生活世界中保持主动与持续？","primary_category":"cs.CL","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM代理","长期任务评测","模拟世界"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10875","has_summary":false},{"id":"2608.12104","title":"No One to Blame: A Framework of Constitutive AI Unaccountability","zh_title":"无人可责：构成性AI不可问责性的框架","primary_category":"cs.CY","date":"2026-08-18","score":0,"bucket":"other","tags":["AI问责制","多智能体系统","社会技术系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12104","has_summary":false},{"id":"2608.13606","title":"MobileMem: Learning from a Year of Mobile Experiences","zh_title":"MobileMem：从一年的移动体验中学习","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["长期记忆","个人助手","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13606","has_summary":false},{"id":"2608.14558","title":"The Unwritten Benchmark: A New Challenge for Multimodal Machine Learning in Abstract Perceptual Reasoning","zh_title":"未写出的基准：多模态机器学习在抽象感知推理中的新挑战","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["多模态基准","感知推理","模型评测"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.14558","has_summary":false},{"id":"2608.14631","title":"Accuracy and Reliability of Large Language Models in Cosmetic Chemistry and Skin Health: A Benchmarking Study","zh_title":"大语言模型在化妆品化学与皮肤健康中的准确性与可靠性：一项基准研究","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM评测","化妆品化学","可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14631","has_summary":false},{"id":"2608.14765","title":"Agentic Data Cleaning Without a Clean Reference: An Experimental Study of Capabilities and Trade-offs","zh_title":"无干净参照的智能体数据清洗：能力与权衡的实验研究","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["数据清洗","LLM agent","多智能体协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14765","has_summary":false},{"id":"2608.14795","title":"Individual Disempowerment through an Advice Channel: Control Loss when Influence is Endogenous","zh_title":"通过建议渠道的个体去权能化：内生影响下的控制损失","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["AI安全","人机交互","控制权"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14795","has_summary":false},{"id":"2608.14804","title":"Generated Context versus Governed State: Functional Conditions for Accountable Longitudinal Clinical Reasoning","zh_title":"生成上下文与受治理状态：可问责纵向临床推理的功能条件","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["临床AI","状态治理","问责框架"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14804","has_summary":false},{"id":"2608.14992","title":"Does a Tool Result Carry More Authority Than Plain Text? Three Prospective Studies of False-Claim Adoption in a Synthetic Assignment Task with Claude Opus 5","zh_title":"工具结果是否比纯文本更具权威性？三项关于Claude Opus 5在合成任务中采纳错误声明的前瞻性研究","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM行为","工具结果权威性","模型评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14992","has_summary":false},{"id":"2608.15131","title":"Platform Adaptation Under Governance Interventions: Actor Best-Response Modeling and an External Public-Case Benchmark","zh_title":"治理干预下的平台适应：行动者最佳响应建模与外部公共案例基准","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["平台治理","多智能体仿真","信息系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15131","has_summary":false},{"id":"2608.15254","title":"Demographic Injection in Medical Language Models under Diversity, Equity, and Inclusion Prompts","zh_title":"多样性、公平与包容提示下医学语言模型的人口统计注入","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM偏差","医疗AI","提示工程"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15254","has_summary":false},{"id":"2608.15309","title":"Physiological World Models for Human State Transitions","zh_title":"用于人类状态转换的生理世界模型","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["生理建模","健康AI","状态转换"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.15309","has_summary":false},{"id":"2608.16370","title":"What Does Context Compression Cost an Agent? Interaction Costs Unrevealed by Task-Completion Metrics","zh_title":"上下文压缩对智能体的代价是什么？任务完成指标未揭示的交互成本","primary_category":"cs.AI","date":"2026-08-18","score":0,"bucket":"other","tags":["上下文压缩","智能体交互成本","工具调用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16370","has_summary":false},{"id":"2608.14577","title":"HarmProfile: Characterizing Harmful Distributions in Frontier LLMs","zh_title":"HarmProfile：刻画前沿大语言模型中的有害分布","primary_category":"cs.CL","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM安全","基准数据集","有害内容"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14577","has_summary":false},{"id":"2608.14609","title":"Understanding AI Anxiety in the Workplace: A Multimethod Investigation Using Fear Acquisition Theory and the Technology Acceptance Model","zh_title":"理解职场中的AI焦虑：基于恐惧习得理论与技术接受模型的多方法研究","primary_category":"cs.CY","date":"2026-08-18","score":0,"bucket":"other","tags":["AI焦虑","人类被试","技术接受模型"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.14609","has_summary":false},{"id":"2608.14625","title":"Local AI pre-screening for human triple-blind peer review in health sciences","zh_title":"健康科学领域人类三盲同行评审的本地AI预筛选","primary_category":"cs.CY","date":"2026-08-18","score":0,"bucket":"other","tags":["同行评审","多LLM框架","AI辅助审稿"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14625","has_summary":false},{"id":"2608.15286","title":"No Task Fails Every Time: Why One-Shot Audits Are Structurally Blind to Agent Damage","zh_title":"没有任务每次都失败：为何一次性审计对智能体损害存在结构性盲区","primary_category":"cs.LG","date":"2026-08-18","score":0,"bucket":"other","tags":["多智能体系统","可靠性评估","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15286","has_summary":false},{"id":"2608.15689","title":"Integrating Persuasion Theory into the Epidemiological Modelling of Health Misinformation Spread on Social Media","zh_title":"将说服理论整合到社交媒体健康错误信息传播的流行病学建模中","primary_category":"cs.SI","date":"2026-08-18","score":0,"bucket":"other","tags":["信息传播模型","流行病学","社交媒体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.15689","has_summary":false},{"id":"2608.15867","title":"Feasible and Novel Synthetic Population Generation with Tabular and Sequential Travel Attributes","zh_title":"具有表格和顺序出行属性的可行且新颖的合成人口生成","primary_category":"cs.LG","date":"2026-08-18","score":0,"bucket":"other","tags":["合成人口","交通需求模型","生成对抗网络"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.15867","has_summary":false},{"id":"2608.16747","title":"Would this change your answer? Evaluating Explanations of LLM Behavior In The Wild with Counterfactual Experiments","zh_title":"这会改变你的答案吗？用反事实实验评估野外LLM行为的解释","primary_category":"cs.LG","date":"2026-08-18","score":0,"bucket":"other","tags":["模型可解释性","反事实实验","LLM行为"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.16747","has_summary":false},{"id":"2608.15550","title":"Adoption of Generative AI in the Workplace: Increasing and Shifting the Balance of Productivity and Communication Activity","zh_title":"生成式AI在工作场所的采用：提高并转移生产力与沟通活动的平衡","primary_category":"cs.HC","date":"2026-08-18","score":0,"bucket":"other","tags":["生成式AI采用","工作场所行为","实证研究"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.15550","has_summary":false},{"id":"2608.14605","title":"Psychological Determinants of Academic Integrity in the Use of Generative AI in Higher Education","zh_title":"高等教育中使用生成式人工智能的学术诚信心理决定因素","primary_category":"cs.CY","date":"2026-08-18","score":0,"bucket":"other","tags":["学术诚信","生成式AI","高等教育"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.14605","has_summary":false},{"id":"2608.16323","title":"Predicting, Evaluating, and Explaining Top Misinformation Spreaders via Archetypal User Behavior","zh_title":"通过原型用户行为预测、评估和解释顶级错误信息传播者","primary_category":"cs.SI","date":"2026-08-18","score":0,"bucket":"other","tags":["错误信息传播","用户行为分析","社交网络"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.16323","has_summary":false},{"id":"2608.14663","title":"In-Context Learning to Assess Built Environment Impacts on Perceived Neighborhood Walkability Among Mobility-impaired Older Adults","zh_title":"上下文学习评估建成环境对行动不便老年人感知邻里步行性的影响","primary_category":"cs.LG","date":"2026-08-18","score":0,"bucket":"other","tags":["表格预测","可解释性","建成环境"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14663","has_summary":false},{"id":"2608.14956","title":"LLM-based Framework for Generating and Verifying Parallel DEVS Statecharts","zh_title":"基于LLM的并行DEVS状态图生成与验证框架","primary_category":"cs.LG","date":"2026-08-18","score":0,"bucket":"other","tags":["LLM辅助建模","形式化验证","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14956","has_summary":false},{"id":"2608.15507","title":"Do Language Models Consistently Encode the Current Year?","zh_title":"语言模型是否一致地编码当前年份？","primary_category":"cs.CL","date":"2026-08-18","score":0,"bucket":"other","tags":["时间推理","模型评测","因果机制"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.15507","has_summary":false},{"id":"2608.14079","title":"The conditional superiority of fast silicon sampling","zh_title":"快速硅采样的条件优越性","primary_category":"cs.CL","date":"2026-08-17","score":10,"bucket":"selected","tags":["硅采样","算法保真度","人类仿真"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.14079","has_summary":true},{"id":"2608.10492","title":"INSIDE the Student's Mind: Jointly Modeling Latent Reasoning and Action in LLM Student Simulators","zh_title":"洞察学生思维：联合建模LLM学生模拟器中的潜在推理与行为","primary_category":"cs.AI","date":"2026-08-17","score":9,"bucket":"selected","tags":["LLM仿真","教育模拟","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.10492","has_summary":true},{"id":"2601.20238","title":"Large Language Models Polarize Ideologically but Moderate Affectively in Online Political Discourse","zh_title":"大语言模型在网络政治话语中加剧意识形态极化但缓和情感极化","primary_category":"econ.GN","date":"2026-08-17","score":8,"bucket":"selected","tags":["LLM仿真","政治极化","人类数据对照"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.20238","has_summary":true},{"id":"2608.13712","title":"Reading Between The Lines: Modeling and Evaluating Behavioral Realism in Legal Simulation","zh_title":"字里行间：法律模拟中行为真实性的建模与评估","primary_category":"cs.CY","date":"2026-08-17","score":8,"bucket":"selected","tags":["LLM仿真","法律模拟","行为真实性"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.13712","has_summary":true},{"id":"2608.13786","title":"Do AI chatbots find what experts would? Effects of model, user role, and sample size on study retrieval for medical questions","zh_title":"AI聊天机器人能否找到专家会找到的研究？模型、用户角色和样本量对医学问题研究检索的影响","primary_category":"cs.IR","date":"2026-08-17","score":7,"bucket":"pending","tags":["LLM仿真","医学信息检索","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.13786","has_summary":true},{"id":"2608.14320","title":"AnchorBench: A Multi-Pathway Benchmark for the Anchoring Effect in LLMs","zh_title":"AnchorBench：LLM锚定效应的多路径基准测试","primary_category":"cs.AI","date":"2026-08-17","score":7,"bucket":"pending","tags":["认知偏差","LLM评估","锚定效应"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2608.14320","has_summary":true},{"id":"2608.14399","title":"Whose doctor does the AI recommend? An algorithm audit of reputation and demographic signals in large language model-assisted physician choice","zh_title":"AI推荐哪位医生？大语言模型辅助医生选择中声誉与人口统计信号的算法审计","primary_category":"cs.CY","date":"2026-08-17","score":7,"bucket":"pending","tags":["LLM仿真","算法审计","医疗决策"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.14399","has_summary":true},{"id":"2608.14113","title":"Search or Chat? Comparing How We Learn About Debated Topics","zh_title":"搜索还是聊天？比较我们如何了解有争议的话题","primary_category":"cs.HC","date":"2026-08-17","score":7,"bucket":"pending","tags":["LLM聊天界面","学习效果","用户研究"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.14113","has_summary":true},{"id":"2608.05246","title":"LUNAR: Benchmarking Personalized Large Language Models on UNiversal User BehAvioR Logs","zh_title":"LUNAR：基于通用用户行为日志的个性化大语言模型基准","primary_category":"cs.AI","date":"2026-08-17","score":6,"bucket":"other","tags":["个性化LLM","行为日志合成","基准测试"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.05246","has_summary":false},{"id":"2608.06549","title":"TradeVerse: A Longitudinal Benchmark of Political Negotiation in International Trade","zh_title":"TradeVerse：国际贸易政治谈判的纵向基准","primary_category":"cs.CL","date":"2026-08-17","score":6,"bucket":"other","tags":["LLM谈判模拟","纵向基准","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06549","has_summary":false},{"id":"2608.13567","title":"Modular Cognitive Architecture Emerges in Large Language Models","zh_title":"大型语言模型中涌现出模块化认知架构","primary_category":"cs.AI","date":"2026-08-17","score":6,"bucket":"other","tags":["LLM认知架构","神经科学类比","模型可解释性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.13567","has_summary":false},{"id":"2608.13563","title":"Proxy-Validated LLM UX Micro-Simulations: An Artifact-First Protocol for Early-Stage Decision Support","zh_title":"代理验证的LLM用户体验微仿真：面向早期决策支持的人工制品优先协议","primary_category":"cs.HC","date":"2026-08-17","score":6,"bucket":"other","tags":["LLM仿真","用户体验","代理验证"],"rubric_hits":["D1","B1"],"abs_url":"https://arxiv.org/abs/2608.13563","has_summary":false},{"id":"2608.13835","title":"When Lexical Change Misleads: Rethinking Dynamic Topic Model Evaluation with Traditional and LLM-Based Metrics","zh_title":"当词汇变化误导：用传统与基于LLM的指标重新思考动态主题模型评估","primary_category":"cs.CL","date":"2026-08-17","score":5,"bucket":"other","tags":["LLM评估","主题模型","语义连贯性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.13835","has_summary":false},{"id":"2608.13840","title":"ASSERT: A Measurement Pipeline for GenAI Audits","zh_title":"ASSERT：生成式AI审计的测量流水线","primary_category":"cs.CL","date":"2026-08-17","score":5,"bucket":"other","tags":["GenAI审计","测量规范","模拟用户"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.13840","has_summary":false},{"id":"2608.13787","title":"From Passive Delegates to Strategic Negotiators: Reinforcing Social Reasoning in Small Language Models with SocialRL","zh_title":"从被动代理到战略谈判者：用SocialRL强化小语言模型的社会推理","primary_category":"cs.AI","date":"2026-08-17","score":5,"bucket":"other","tags":["多智能体谈判","社会推理","强化学习"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.13787","has_summary":false},{"id":"2608.13921","title":"When Personal Memory Has No Single Answer: Evaluating LLM Agents under Irreducible Conflict","zh_title":"当个人记忆没有单一答案：在不可约冲突下评估LLM智能体","primary_category":"cs.AI","date":"2026-08-17","score":5,"bucket":"other","tags":["LLM记忆","冲突处理","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.13921","has_summary":false},{"id":"2608.14161","title":"BiasTrace: Linking Reasoning Behaviours to Biased Outputs in LLMs","zh_title":"BiasTrace：将推理行为与LLM中的偏见输出联系起来","primary_category":"cs.AI","date":"2026-08-17","score":5,"bucket":"other","tags":["LLM偏见","推理行为","标注方案"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.14161","has_summary":false},{"id":"2608.07852","title":"\"Many Are My Names\": The Anatomy of the Assistant and Its Personas via Sparse Autoencoders","zh_title":"“我名众多”：通过稀疏自编码器剖析助手及其人格面具的解剖结构","primary_category":"cs.CL","date":"2026-08-17","score":2,"bucket":"other","tags":["角色扮演","模型可解释性","说话者表征"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07852","has_summary":false},{"id":"2608.12831","title":"Fast A/B/n Testing: Exact Multi-Policy Comparison via Tree-Coupled Feedback Sharing","zh_title":"快速A/B/n测试：通过树耦合反馈共享进行精确多策略比较","primary_category":"cs.LG","date":"2026-08-17","score":2,"bucket":"other","tags":["A/B测试","多臂老虎机","统计方法"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12831","has_summary":false},{"id":"2608.13760","title":"Amplified Does Not Mean Predictive: Reasoning Behaviors in Thinking Models","zh_title":"放大不等于预测：思维模型中的推理行为","primary_category":"cs.CL","date":"2026-08-17","score":2,"bucket":"other","tags":["推理模型","行为分析","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13760","has_summary":false},{"id":"2608.13604","title":"Cross-Disciplinary Taxonomy and Modeling of Misunderstanding Generation, Amplification, and Detection, from Pragmatics to AI Agents","zh_title":"跨学科误解生成、放大与检测的分类与建模：从语用学到AI智能体","primary_category":"cs.AI","date":"2026-08-17","score":2,"bucket":"other","tags":["误解检测","多智能体通信","语用学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13604","has_summary":false},{"id":"2608.13866","title":"Geometric Filtering of LLM-Generated Samples for Few-Shot Text Classification","zh_title":"用于少样本文本分类的LLM生成样本几何过滤","primary_category":"cs.LG","date":"2026-08-17","score":2,"bucket":"other","tags":["数据增强","文本分类","合成数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13866","has_summary":false},{"id":"2608.14329","title":"A Four-Axis Trustworthiness Benchmark for LLM-as-Judge in Principle-Based Regulation","zh_title":"基于原则监管中LLM作为评判者的四轴可信度基准","primary_category":"cs.CR","date":"2026-08-17","score":2,"bucket":"other","tags":["LLM评估","监管合规","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14329","has_summary":false},{"id":"2608.14522","title":"Participatory Moral AI Is Not Neutral: The Invisible Hand of Developers","zh_title":"参与式道德AI并非中立：开发者的无形之手","primary_category":"cs.AI","date":"2026-08-17","score":2,"bucket":"other","tags":["道德AI","偏好聚合","人类实验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.14522","has_summary":false},{"id":"2608.14528","title":"Handover of In-Context Learning State Across Session Boundaries","zh_title":"跨会话边界的上下文学习状态交接","primary_category":"cs.AI","date":"2026-08-17","score":2,"bucket":"other","tags":["LLM会话交接","上下文学习","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14528","has_summary":false},{"id":"2608.13944","title":"Musical Mirrors: The LLM as Sounding Board in Songwriting","zh_title":"音乐之镜：LLM作为歌曲创作中的共鸣板","primary_category":"cs.HC","date":"2026-08-17","score":2,"bucket":"other","tags":["人机交互","创意实践","LLM应用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.13944","has_summary":false},{"id":"2608.14130","title":"AlignFace: Human-Aligned Face Similarity Metric with Interpretable Concept Relations","zh_title":"AlignFace：具有可解释概念关系的人脸相似度度量","primary_category":"cs.MM","date":"2026-08-17","score":2,"bucket":"other","tags":["人脸相似度","感知评估","计算机视觉"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.14130","has_summary":false},{"id":"2608.14093","title":"AppLooper: An Agentic Application Engineering Loop for Accountable Release with Virtual-User Feedback","zh_title":"AppLooper：一种面向可问责发布、结合虚拟用户反馈的智能体应用工程循环","primary_category":"cs.HC","date":"2026-08-17","score":2,"bucket":"other","tags":["多智能体系统","应用工程","虚拟用户测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14093","has_summary":false},{"id":"2608.11625","title":"Making AI-Generated Feedback Matter: A Large-Scale Study of Feedback Workflows and Student Enactment","zh_title":"让AI生成的反馈发挥作用：反馈工作流与学生实施的大规模研究","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["AI反馈","教育技术","学习分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11625","has_summary":false},{"id":"2608.13624","title":"Measuring Fairness in Large Audio Language Models via Semantic-Aware Bias Estimation","zh_title":"通过语义感知偏差估计测量大型音频语言模型中的公平性","primary_category":"cs.CL","date":"2026-08-17","score":0,"bucket":"other","tags":["公平性评估","音频语言模型","偏差估计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13624","has_summary":false},{"id":"2608.14210","title":"How Much Do Legal RAG Systems Still Hallucinate?","zh_title":"法律RAG系统仍会产生多少幻觉？","primary_category":"cs.CL","date":"2026-08-17","score":0,"bucket":"other","tags":["法律RAG","幻觉评估","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14210","has_summary":false},{"id":"2608.13674","title":"Asymmetric Discourse Homogenization and Shared Language Technology: Evidence from Reddit","zh_title":"不对称话语同质化与共享语言技术：来自Reddit的证据","primary_category":"cs.CY","date":"2026-08-17","score":0,"bucket":"other","tags":["社会计算","政治话语","Reddit"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13674","has_summary":false},{"id":"2608.14198","title":"MINT: A Universal Zero-Shot Predictor for Transaction Data","zh_title":"MINT：交易数据的通用零样本预测器","primary_category":"cs.LG","date":"2026-08-17","score":0,"bucket":"other","tags":["金融预测","多模态LLM","零样本学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14198","has_summary":false},{"id":"2608.14286","title":"Seeing Red, Thinking Bad: Color Bias in Vision Language Models","zh_title":"见红思坏：视觉语言模型中的颜色偏差","primary_category":"cs.CV","date":"2026-08-17","score":0,"bucket":"other","tags":["视觉语言模型","偏见分析","对抗性提示"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14286","has_summary":false},{"id":"2608.14509","title":"Split the Labor: Separating Evidence Interpretation from Decision Aggregation","zh_title":"分工：将证据解释与决策聚合分离","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["多源信息聚合","决策系统","语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14509","has_summary":false},{"id":"2608.13577","title":"AI Evaluation Should Work With Humans","zh_title":"AI评估应与人类合作","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["AI评估","人机协作","立场论文"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13577","has_summary":false},{"id":"2608.14014","title":"Buy the Rumor, Sell the News: When Is News Priced In?","zh_title":"买谣言，卖新闻：新闻何时被定价？","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["金融新闻","LLM分类","市场效率"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.14014","has_summary":false},{"id":"2608.14152","title":"Towards Efficient Multimodal and Multilingual Opinion Extraction for STI: A QLoRA-Based Fine-Tuning Approach","zh_title":"面向科技情报的高效多模态多语言观点抽取：基于QLoRA的微调方法","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["观点抽取","多模态","科技情报"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.14152","has_summary":false},{"id":"2608.14179","title":"Can Language Models Understand mmWave Data? Benchmarking Large Language Models for mmWave Radar-Based Human Understanding","zh_title":"语言模型能理解毫米波数据吗？基于毫米波雷达的人类理解基准测试","primary_category":"cs.AI","date":"2026-08-17","score":0,"bucket":"other","tags":["毫米波雷达","多模态问答","感知基准"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.14179","has_summary":false},{"id":"2608.14132","title":"Act2Intention: A Benchmark For Developing Active Mobile Agents Through Inferring User Intention from GUI Actions","zh_title":"Act2Intention：通过从GUI动作推断用户意图来开发主动移动智能体的基准","primary_category":"cs.HC","date":"2026-08-17","score":0,"bucket":"other","tags":["移动智能体","意图理解","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.14132","has_summary":false},{"id":"2608.14291","title":"Human and Artificial Intelligence - Promoting Trustworthy and Understandable Collaboration","zh_title":"人类与人工智能——促进可信赖且可理解的协作","primary_category":"cs.HC","date":"2026-08-17","score":0,"bucket":"other","tags":["人机交互","可解释性","用户调查"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.14291","has_summary":false},{"id":"2608.13864","title":"Audience capture, selective exposure, affective assimilation or ideological sorting? Polarisation of climate politics under low media-party parallelism","zh_title":"低媒体-政党平行主义下气候政治的两极分化：受众捕获、选择性接触、情感同化还是意识形态排序？","primary_category":"cs.SI","date":"2026-08-17","score":0,"bucket":"other","tags":["社交媒体分析","政治极化","计算社会科学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13864","has_summary":false},{"id":"2608.14141","title":"Who Owns the Online Media?","zh_title":"谁拥有在线媒体？","primary_category":"econ.GN","date":"2026-08-17","score":0,"bucket":"other","tags":["媒体所有权","网络分析","内容相似性"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.14141","has_summary":false},{"id":"2608.12368","title":"Agreement Is Not Alignment: Divergent Moral Grounds in Human and LLM Ethical Judgments","zh_title":"一致不等于对齐：人类与LLM道德判断中分歧的道德依据","primary_category":"cs.AI","date":"2026-08-15","score":8,"bucket":"selected","tags":["LLM对齐评估","道德判断","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.12368","has_summary":true},{"id":"2608.12788","title":"ARAC: Benchmarking Auto-Research's Alignment and Completeness on End-to-End Researchs","zh_title":"ARAC：基准测试自动研究在端到端研究中的对齐性与完整性","primary_category":"cs.AI","date":"2026-08-15","score":7,"bucket":"pending","tags":["自动研究评估","人类行为对齐","基准测试"],"rubric_hits":["A4","B1"],"abs_url":"https://arxiv.org/abs/2608.12788","has_summary":true},{"id":"2608.12387","title":"Query Timing Produces Opposite Positional Biases Between LLMs and Humans","zh_title":"查询时机导致LLM与人类之间相反的位置偏差","primary_category":"cs.CL","date":"2026-08-15","score":7,"bucket":"pending","tags":["LLM偏差","人类对照","认知建模"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.12387","has_summary":true},{"id":"2608.12630","title":"Novels generated by language models show compressed formal variation","zh_title":"语言模型生成的小说显示出压缩的形式变异","primary_category":"cs.CL","date":"2026-08-15","score":6,"bucket":"other","tags":["LLM生成","文本变异","人类对比"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.12630","has_summary":false},{"id":"2608.09164","title":"CIDER: A Dataset of Contextual Disclosure Boundaries for Privacy Preference Alignment","zh_title":"CIDER：用于隐私偏好对齐的情境披露边界数据集","primary_category":"cs.AI","date":"2026-08-15","score":5,"bucket":"other","tags":["隐私偏好","个性化对齐","数据集"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.09164","has_summary":false},{"id":"2608.12373","title":"Don't Want Your LLM to Recommend Nuclear Strike? Try Asking It in Japanese","zh_title":"不想让大语言模型推荐核打击？试试用日语提问","primary_category":"cs.AI","date":"2026-08-15","score":5,"bucket":"other","tags":["LLM安全对齐","跨语言行为","模型测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.12373","has_summary":false},{"id":"2608.13258","title":"Self-Referential Induction Increases Response Instability Relative to Unresolvable and Verifiable Questions in Large Language Models","zh_title":"自指诱导增加大语言模型回答不稳定性：与不可解问题和可验证问题的比较","primary_category":"cs.CL","date":"2026-08-15","score":5,"bucket":"other","tags":["LLM主观报告","回答稳定性","自指提示"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.13258","has_summary":false},{"id":"2608.12345","title":"Diagnostic Foundation for Evaluating LLMs' Research Integrity as Co-Scientists","zh_title":"评估LLM作为共同科学家研究诚信的诊断基础","primary_category":"cs.AI","date":"2026-08-15","score":2,"bucket":"other","tags":["LLM评测","科研诚信","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.12345","has_summary":false},{"id":"2608.13046","title":"BoardroomAI: Dependency-Aware Human-Steerable Multi-Agent Deliberation through Evolving Decision Graphs","zh_title":"BoardroomAI：通过演化决策图实现依赖感知的人类可操控多智能体审议","primary_category":"cs.AI","date":"2026-08-15","score":2,"bucket":"other","tags":["多智能体系统","人机协作","决策图"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13046","has_summary":false},{"id":"2608.13069","title":"Behavioral Reprogramming of Open-Weights Models: Cognitive Plasticity and Alignment Bounds","zh_title":"开放权重模型的行为重编程：认知可塑性与对齐边界","primary_category":"cs.AI","date":"2026-08-15","score":2,"bucket":"other","tags":["行为重编程","角色扮演","模型对齐"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.13069","has_summary":false},{"id":"2608.13120","title":"SkillEvo: Self-Renewing Evolution Gradients from Multi-Turn Interaction Feedback","zh_title":"SkillEvo：从多轮交互反馈中自我更新的进化梯度","primary_category":"cs.AI","date":"2026-08-15","score":2,"bucket":"other","tags":["多智能体系统","技能进化","反馈生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13120","has_summary":false},{"id":"2608.12377","title":"From Observation to Intervention: Memory in Brains and Large Language Models","zh_title":"从观察到干预：大脑与大语言模型中的记忆","primary_category":"q-bio.NC","date":"2026-08-15","score":2,"bucket":"other","tags":["记忆机制","神经科学","LLM可解释性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12377","has_summary":false},{"id":"2608.13328","title":"It's How You Ask: Gender-Associated Linguistic Bias in LLMs","zh_title":"如何提问：大语言模型中与性别相关的语言偏见","primary_category":"cs.CL","date":"2026-08-15","score":2,"bucket":"other","tags":["语言偏见","模型行为","性别差异"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13328","has_summary":false},{"id":"2608.12344","title":"Predicting consumer-technology ownership without a diffusion history","zh_title":"无扩散历史下预测消费者技术拥有率","primary_category":"cs.CL","date":"2026-08-14","score":9,"bucket":"selected","tags":["LLM仿真","消费者行为","算法保真度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.12344","has_summary":true},{"id":"2608.12339","title":"Mimicry without understanding: the origins of decision bias in large language models","zh_title":"无理解的模仿：大语言模型中决策偏差的起源","primary_category":"cs.CL","date":"2026-08-14","score":8,"bucket":"selected","tags":["LLM偏差","经济决策","仿真可靠性"],"rubric_hits":["A2","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.12339","has_summary":true},{"id":"2608.13454","title":"Before You Say It: Anticipating Verbal Behavior from Longitudinal Everyday Conversations with LLMs","zh_title":"在你说出口之前：利用大语言模型从纵向日常对话中预测言语行为","primary_category":"cs.HC","date":"2026-08-14","score":7,"bucket":"pending","tags":["LLM行为预测","纵向对话数据","个性化建模"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.13454","has_summary":true},{"id":"2608.12352","title":"Why AI Governance Frameworks Are Hard to Adopt: A Role-Based Stress Test of the NIST AI RMF","zh_title":"为何AI治理框架难以采纳：对NIST AI RMF的基于角色的压力测试","primary_category":"cs.CY","date":"2026-08-14","score":7,"bucket":"pending","tags":["LLM角色模拟","AI治理","压力测试"],"rubric_hits":["A1","B4"],"abs_url":"https://arxiv.org/abs/2608.12352","has_summary":true},{"id":"2608.12717","title":"Perturbation-based Regional Interpretability through Subtraction Mapping (PRISM): naming-error dissociations in language models and post-stroke aphasia","zh_title":"基于扰动区域可解释性的减法映射（PRISM）：语言模型与卒中后失语症的命名错误分离","primary_category":"cs.LG","date":"2026-08-14","score":7,"bucket":"pending","tags":["LLM可解释性","神经语言学","人类对照"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.12717","has_summary":true},{"id":"2608.03421","title":"When Truth Is Distributed: Misinformation Derails Collective Fact Recovery in LLM-Based Multi-Agent Systems","zh_title":"当真相被分散：错误信息在基于LLM的多智能体系统中破坏集体事实恢复","primary_category":"cs.MA","date":"2026-08-14","score":5,"bucket":"other","tags":["多智能体系统","信息传播","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.03421","has_summary":false},{"id":"2608.12358","title":"Interaction Readiness: A Framework for Building and Evaluating AI Agents in Human Roles","zh_title":"交互就绪：构建和评估人类角色AI代理的框架","primary_category":"cs.HC","date":"2026-08-14","score":5,"bucket":"other","tags":["AI代理评估","角色扮演","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.12358","has_summary":false},{"id":"2608.12582","title":"Not All Nudges Land: Behavioral Controllability and Elaboration Quality in AI-Supported Journaling","zh_title":"并非所有助推都有效：AI辅助日记中的行为可控性与阐述质量","primary_category":"cs.HC","date":"2026-08-14","score":5,"bucket":"other","tags":["LLM标注","行为改变","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.12582","has_summary":false},{"id":"2608.13017","title":"How LLMs Respond to Escalating Delusions: Four Longitudinal Trajectories of Model Behavior","zh_title":"LLM如何应对升级性妄想：模型行为的四种纵向轨迹","primary_category":"cs.HC","date":"2026-08-14","score":5,"bucket":"other","tags":["LLM安全","精神病学","纵向评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.13017","has_summary":false},{"id":"2608.12329","title":"AnchorSIPS: A Synthetic Dataset and Evaluation Resource for Evidence-Supported Psychosis-Risk Symptom Measurement","zh_title":"AnchorSIPS：用于证据支持的精神病风险症状测量的合成数据集与评估资源","primary_category":"cs.CL","date":"2026-08-14","score":5,"bucket":"other","tags":["合成数据","精神病风险评估","LLM生成"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.12329","has_summary":false},{"id":"2608.12547","title":"Do LLMs Beat Nash? Testing Decentralized Coordination in Self-Play Multi-Agent Games","zh_title":"LLM能否超越纳什均衡？在自博弈多智能体游戏中测试去中心化协调","primary_category":"cs.MA","date":"2026-08-14","score":3,"bucket":"other","tags":["多智能体系统","博弈论","自博弈"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12547","has_summary":false},{"id":"2608.07516","title":"Catch the Patient, Not the AI: Collective Sensemaking in an Online Health Community","zh_title":"抓住患者，而非AI：在线健康社区中的集体意义建构","primary_category":"cs.HC","date":"2026-08-14","score":2,"bucket":"other","tags":["在线健康社区","AI工具使用","集体意义建构"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07516","has_summary":false},{"id":"2608.11955","title":"Philosophical vertigo with artificial intelligence","zh_title":"人工智能引发的哲学眩晕","primary_category":"cs.CY","date":"2026-08-14","score":2,"bucket":"other","tags":["人机交互","哲学影响","AI安全"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11955","has_summary":false},{"id":"2608.12324","title":"When AI Is Your Pastor: A Benchmark for Theological Triage and Pastoral Guidance in Large Language Models","zh_title":"当AI成为你的牧师：大语言模型神学分类与教牧指导基准","primary_category":"cs.CY","date":"2026-08-14","score":2,"bucket":"other","tags":["LLM评估","宗教咨询","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12324","has_summary":false},{"id":"2608.13250","title":"Follow the Norm: Accounting for Fine-Tuning and Prompt Effects on Model Rationales","zh_title":"遵循规范：考虑微调和提示对模型理由的影响","primary_category":"cs.CY","date":"2026-08-14","score":2,"bucket":"other","tags":["AI对齐","规范推理","模型行为"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.13250","has_summary":false},{"id":"2608.13369","title":"Credible, Not Always Correct: How Reddit Users Verify AI-Generated Legal Advice","zh_title":"可信但不总是正确：Reddit用户如何验证AI生成的法律建议","primary_category":"cs.CY","date":"2026-08-14","score":2,"bucket":"other","tags":["AI法律建议","用户行为","可信度"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.13369","has_summary":false},{"id":"2608.12372","title":"Position: We Need Practical AI Alignment Methods to Mirror Human Reasoning","zh_title":"立场：我们需要实用的AI对齐方法来反映人类推理","primary_category":"cs.AI","date":"2026-08-14","score":2,"bucket":"other","tags":["AI对齐","认知对齐","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12372","has_summary":false},{"id":"2608.13482","title":"Synthetic Persona Pretraining: Alignment from Token Zero","zh_title":"合成人格预训练：从零开始的对齐","primary_category":"cs.LG","date":"2026-08-14","score":2,"bucket":"other","tags":["模型对齐","预训练","合成数据"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.13482","has_summary":false},{"id":"2608.13063","title":"Explanatory Engagement Under Rare Anomalous Failure: Asymptotic Rarity in Model Behavior (or: The Asymptotic AI)","zh_title":"罕见异常失败下的解释性参与：模型行为中的渐近稀有性（或：渐近AI）","primary_category":"cs.AI","date":"2026-08-14","score":2,"bucket":"other","tags":["LLM行为分析","异常检测","工具调用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13063","has_summary":false},{"id":"2608.13267","title":"How Do VLMs Behave When Blind or Misled? Behavioral Evaluation of VLMs on Scientific Figures","zh_title":"视觉语言模型在失明或被误导时如何表现？科学图表上的行为评估","primary_category":"cs.CL","date":"2026-08-14","score":2,"bucket":"other","tags":["VLM评测","科学图表理解","行为可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13267","has_summary":false},{"id":"2608.13510","title":"On the Structural Limits of Machine Learning Decision Systems: An Information-Theoretic, Interaction-Based, and Stochastic-Dynamical Perspective","zh_title":"机器学习决策系统的结构极限：信息论、交互与随机动力学视角","primary_category":"math.ST","date":"2026-08-14","score":2,"bucket":"other","tags":["信息论","机器学习理论","决策系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13510","has_summary":false},{"id":"2608.07070","title":"Coordinated incentives in AI-generated misinformation governance","zh_title":"AI生成虚假信息治理中的协调激励","primary_category":"physics.soc-ph","date":"2026-08-14","score":0,"bucket":"other","tags":["演化博弈","虚假信息治理","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07070","has_summary":false},{"id":"2608.08443","title":"Private Etymology: Designing Relational Reuse of Shared Symbols in Long-Term Human-AI Interaction","zh_title":"私人词源：设计长期人机交互中共享符号的关系性重用","primary_category":"cs.HC","date":"2026-08-14","score":0,"bucket":"other","tags":["人机交互","对话系统","共享符号"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.08443","has_summary":false},{"id":"2608.09959","title":"AIFS-TC: A simple correction competitive with the operational frontier for tropical cyclone intensity forecasting","zh_title":"AIFS-TC：一种与业务前沿竞争的热带气旋强度预报简单修正方法","primary_category":"physics.ao-ph","date":"2026-08-14","score":0,"bucket":"other","tags":["气象预报","LLM辅助开发","热带气旋"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.09959","has_summary":false},{"id":"2608.11392","title":"AI Guardrail Survival under Single-Cycle Agentic Self-Summarization","zh_title":"单周期智能体自摘要下的AI护栏存活性","primary_category":"cs.CR","date":"2026-08-14","score":0,"bucket":"other","tags":["AI安全","多智能体系统","上下文压缩"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11392","has_summary":false},{"id":"2608.12895","title":"Agent Behavioral Contracts II: Certifying Compositional Reliability Without Assuming Independence","zh_title":"智能体行为契约 II：在不假设独立性的情况下认证组合可靠性","primary_category":"cs.AI","date":"2026-08-14","score":0,"bucket":"other","tags":["多智能体系统","可靠性认证","统计依赖"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12895","has_summary":false},{"id":"2608.13136","title":"LigBench: A Unified and Human-Aligned Benchmark for LLM-based Research Idea Generation","zh_title":"LigBench：面向基于LLM的研究想法生成的统一且与人类对齐的基准","primary_category":"cs.CL","date":"2026-08-14","score":0,"bucket":"other","tags":["LLM评测","研究想法生成","基准数据集"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13136","has_summary":false},{"id":"2608.13329","title":"A Probe Direction Is a Property of Its Prompt","zh_title":"探针方向是其提示的属性","primary_category":"cs.LG","date":"2026-08-14","score":0,"bucket":"other","tags":["模型评估","提示敏感性","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13329","has_summary":false},{"id":"2608.12444","title":"Non-Degenerate Risk Certification for Automated Security Decisions: A Decision-Contract Theory with ATT\\&CK-Aligned Triage as a Worked Instance","zh_title":"自动化安全决策的非退化风险认证：以ATT&CK对齐的分类为实例的决策契约理论","primary_category":"cs.CR","date":"2026-08-14","score":0,"bucket":"other","tags":["LLM安全决策","风险认证","入侵检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.12444","has_summary":false},{"id":"2608.13167","title":"TRAPSBench: Vision-Language Models Encode but Fail to Express Epistemic Restraint","zh_title":"TRAPSBench：视觉语言模型编码了认知克制但未能表达","primary_category":"cs.CV","date":"2026-08-14","score":0,"bucket":"other","tags":["视觉语言模型","模型评测","弃权行为"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.13167","has_summary":false},{"id":"2608.13315","title":"Keep, Customize, or Exit: Default Design and Token Pricing in LLM Reasoning Services","zh_title":"保留、定制或退出：LLM推理服务中的默认设计与代币定价","primary_category":"cs.GT","date":"2026-08-14","score":0,"bucket":"other","tags":["LLM服务定价","博弈论","推理服务"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.13315","has_summary":false},{"id":"2608.11794","title":"Toward Meaningful Transparency for AI Chatbots: Disclosing Persuasive Intent Reduces Persuasion","zh_title":"面向AI聊天机器人的有意义透明度：披露说服意图可降低说服效果","primary_category":"cs.CY","date":"2026-08-13","score":8,"bucket":"selected","tags":["LLM仿真","说服实验","透明度"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.11794","has_summary":true},{"id":"2608.11528","title":"Group Alignment-Induced Sycophancy: A Two-Sided Evaluation of Steerable Pluralistic Alignment","zh_title":"群体对齐引发的谄媚：可引导多元对齐的双面评估","primary_category":"cs.CL","date":"2026-08-13","score":7,"bucket":"pending","tags":["LLM仿真","群体对齐","谄媚偏差"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.11528","has_summary":true},{"id":"2608.11215","title":"Poor Man's Agentic Modeling: Simulating Large LLM-Agent Societies on a Laptop","zh_title":"穷人的代理建模：在笔记本电脑上模拟大型LLM代理社会","primary_category":"cs.AI","date":"2026-08-13","score":7,"bucket":"pending","tags":["LLM代理社会模拟","计算社会科学","代理建模"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2608.11215","has_summary":true},{"id":"2608.11493","title":"From Prompting to Behavioral Alignment: Personalized LLM Judges for Recommendation Evaluation","zh_title":"从提示到行为对齐：用于推荐评估的个性化LLM评判器","primary_category":"cs.AI","date":"2026-08-13","score":7,"bucket":"pending","tags":["LLM仿真","行为对齐","推荐评估"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.11493","has_summary":true},{"id":"2608.11510","title":"Conflict and Congruency Effects in Large Language Models: In-Weight and In-Context Competition in a Verbal Conflict Task","zh_title":"大语言模型中的冲突与一致性效应：言语冲突任务中的权重内与上下文内竞争","primary_category":"q-bio.NC","date":"2026-08-13","score":7,"bucket":"pending","tags":["LLM认知机制","冲突任务","人类对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.11510","has_summary":true},{"id":"2608.11008","title":"Templated or fully synthetic? Prompt construction as a confound in measuring LLM political stance beyond writing assistance","zh_title":"模板化还是完全合成？提示构建作为测量LLM政治立场超越写作辅助的混淆因素","primary_category":"cs.CL","date":"2026-08-13","score":6,"bucket":"other","tags":["LLM立场测量","提示构建","生态效度"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.11008","has_summary":false},{"id":"2608.11460","title":"Principal Trait Analysis: Towards Deriving \"Skills\" in Human-AI Collaboration","zh_title":"主特质分析：推导人机协作中的“技能”","primary_category":"cs.CL","date":"2026-08-13","score":6,"bucket":"other","tags":["人机协作","对话分析","技能提取"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.11460","has_summary":false},{"id":"2608.11649","title":"Who Would You Vote For? Auditing Political Alignment in LLMs: An Italian Case-Study","zh_title":"你会投票给谁？审计大语言模型的政治倾向：意大利案例研究","primary_category":"cs.CL","date":"2026-08-13","score":6,"bucket":"other","tags":["LLM政治倾向","审计框架","提示敏感性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.11649","has_summary":false},{"id":"2608.12125","title":"Do LLMs Take Care of Their Own? Similarity Signals Can Induce Cooperation","zh_title":"LLM 会照顾自己人吗？相似性信号可诱导合作","primary_category":"cs.GT","date":"2026-08-13","score":6,"bucket":"other","tags":["LLM 博弈","社会模拟","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.12125","has_summary":false},{"id":"2608.11225","title":"Identity from the Outside: A Conceptual Framework and Research Program for AI Personality Clones","zh_title":"从外部看身份：AI人格克隆的概念框架与研究计划","primary_category":"cs.AI","date":"2026-08-13","score":6,"bucket":"other","tags":["AI人格克隆","身份仿真","评估框架"],"rubric_hits":["D2","A4"],"abs_url":"https://arxiv.org/abs/2608.11225","has_summary":false},{"id":"2608.11624","title":"Learning to Persuade Exposes How Easily LLMs Abandon Correct Beliefs","zh_title":"学习说服暴露了LLM多么容易放弃正确信念","primary_category":"cs.CL","date":"2026-08-13","score":5,"bucket":"other","tags":["LLM信念","对抗性说服","模型脆弱性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.11624","has_summary":false},{"id":"2608.11735","title":"Locating and Controlling Implicit Personalization in Large Language Models","zh_title":"定位与控制大语言模型中的隐式个性化","primary_category":"cs.CL","date":"2026-08-13","score":5,"bucket":"other","tags":["LLM偏见","内部表征","因果干预"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.11735","has_summary":false},{"id":"2608.11207","title":"Dynamic Governance of Multi-LLM Agent Systems for Collaborative Conversational Outcomes","zh_title":"多LLM智能体系统的动态治理以实现协作对话结果","primary_category":"cs.AI","date":"2026-08-13","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","对话治理"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.11207","has_summary":false},{"id":"2608.11552","title":"Beyond Single-Turn Confidence: Trajectory-Adapted Uncertainty Quantification for LLM Agents","zh_title":"超越单轮置信度：面向LLM智能体的轨迹自适应不确定性量化","primary_category":"cs.CL","date":"2026-08-13","score":2,"bucket":"other","tags":["不确定性量化","LLM智能体","工具使用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11552","has_summary":false},{"id":"2608.11694","title":"The Wording Effect: Quantifying Two-Way Drift in LLM Benchmark Performance","zh_title":"措辞效应：量化LLM基准性能的双向漂移","primary_category":"cs.CL","date":"2026-08-13","score":2,"bucket":"other","tags":["基准测试","措辞敏感性","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11694","has_summary":false},{"id":"2608.11513","title":"Do Influence Tactics Matter? Investigating Prompt Framing Effects in LLM Code Generation","zh_title":"影响策略重要吗？探究LLM代码生成中的提示框架效应","primary_category":"cs.SE","date":"2026-08-13","score":2,"bucket":"other","tags":["提示工程","代码生成","LLM评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11513","has_summary":false},{"id":"2608.11247","title":"Conformity Mitigations in Large Language Models Lie on a Single Resistance-Receptivity Frontier","zh_title":"大语言模型中的从众缓解位于单一抵抗-接受前沿","primary_category":"cs.AI","date":"2026-08-13","score":2,"bucket":"other","tags":["多智能体协作","从众行为","模型鲁棒性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11247","has_summary":false},{"id":"2608.11381","title":"From Numbers to Judgment: Specialist LLM Agents and Reinforcement Learning for European Listed Real Estate","zh_title":"从数字到判断：面向欧洲上市房地产的专业LLM智能体与强化学习","primary_category":"cs.AI","date":"2026-08-13","score":2,"bucket":"other","tags":["多智能体系统","金融分析","LLM专业化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11381","has_summary":false},{"id":"2608.11705","title":"Making Your LLMs More Objective: Stabilizing LLM Safety Behavior Across Traits with Trait-Invariant Safety Tuning","zh_title":"让你的大语言模型更客观：通过特质不变安全调优稳定跨特质的安全行为","primary_category":"cs.AI","date":"2026-08-13","score":2,"bucket":"other","tags":["LLM安全","特质不变性","模型行为分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11705","has_summary":false},{"id":"2608.11259","title":"Methodologies for Improving the Quality of AI Tutoring in K-12 Education","zh_title":"改进K-12教育中AI辅导质量的方法论","primary_category":"cs.CY","date":"2026-08-13","score":2,"bucket":"other","tags":["AI辅导","教育技术","LLM应用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11259","has_summary":false},{"id":"2608.11415","title":"TRACES: A Benchmark for Epistemic Reliability in Scientific Reasoning by LLMs","zh_title":"TRACES：评估大语言模型科学推理中认知可靠性的基准","primary_category":"cs.IR","date":"2026-08-13","score":2,"bucket":"other","tags":["LLM评测","科学推理","可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11415","has_summary":false},{"id":"2608.12236","title":"How Organizations Use AI: Evidence from ChatGPT","zh_title":"组织如何使用AI：来自ChatGPT的证据","primary_category":"econ.GN","date":"2026-08-13","score":2,"bucket":"other","tags":["AI采用","企业数据","实证研究"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.12236","has_summary":false},{"id":"2608.11512","title":"Cheap, Fallible Cognition and the Political Economy of Expertise","zh_title":"廉价易错的认知与专业知识的政治经济学","primary_category":"cs.CY","date":"2026-08-13","score":2,"bucket":"other","tags":["AI与劳动力市场","制度分析","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.11512","has_summary":false},{"id":"2608.12292","title":"Teaching a Large Language Model Tutor to Withhold the Answer: A Supervisor Architecture and an Evidence-Driven Method for Tuning Socratic Behavior","zh_title":"教大语言模型导师保留答案：监督者架构与基于证据的苏格拉底行为调优方法","primary_category":"cs.CY","date":"2026-08-13","score":2,"bucket":"other","tags":["智能辅导系统","LLM行为约束","教育技术"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12292","has_summary":false},{"id":"2608.05690","title":"ASIDE: From Conflict Participants to Co-Observers Through Dyadic Spectator Reflection","zh_title":"ASIDE：通过二元旁观反思从冲突参与者转变为共同观察者","primary_category":"cs.HC","date":"2026-08-13","score":0,"bucket":"other","tags":["人机交互","冲突反思","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.05690","has_summary":false},{"id":"2608.09507","title":"Learning Preference Adaptation for Large Language Model Personalization via Verbal Reinforcement Learning","zh_title":"通过语言强化学习实现大语言模型个性化的偏好适配学习","primary_category":"cs.CL","date":"2026-08-13","score":0,"bucket":"other","tags":["LLM个性化","偏好适配","元学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09507","has_summary":false},{"id":"2608.10400","title":"Do Judges Behave Like Algorithms?","zh_title":"法官的行为是否像算法？","primary_category":"cs.LG","date":"2026-08-13","score":0,"bucket":"other","tags":["司法决策","算法可解释性","机器学习"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.10400","has_summary":false},{"id":"2608.11229","title":"Synchronizing Beliefs with Second-Order Theory-of-Mind in Human-Autonomy Teams (Extended Version)","zh_title":"在人机自主团队中用二阶心智理论同步信念（扩展版）","primary_category":"cs.AI","date":"2026-08-13","score":0,"bucket":"other","tags":["机器人学习","偏好学习","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.11229","has_summary":false},{"id":"2608.11245","title":"Towards Sustainable Learning in Online Education: A Reinforcement Learning Approach","zh_title":"迈向在线教育中的可持续学习：一种强化学习方法","primary_category":"cs.AI","date":"2026-08-13","score":0,"bucket":"other","tags":["强化学习","在线教育","个性化学习"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.11245","has_summary":false},{"id":"2608.12097","title":"Graph-Structured Rubrics: Compiling Rubrics into Typed Evaluation Graphs for LLM Judges","zh_title":"图结构评分标准：将评分标准编译为类型化评估图用于LLM评判器","primary_category":"cs.AI","date":"2026-08-13","score":0,"bucket":"other","tags":["LLM评估","评分标准","图结构"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.12097","has_summary":false},{"id":"2509.16749","title":"Evaluating LLM Generated Detection Rules in Cybersecurity","zh_title":"评估大语言模型生成的网络安全检测规则","primary_category":"cs.CR","date":"2026-08-13","score":0,"bucket":"other","tags":["网络安全","LLM评估","规则生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2509.16749","has_summary":false},{"id":"2608.11283","title":"Chemically Meaningful Textualization Enables Explainable Validation of Metal-Organic Frameworks by Large Language Models","zh_title":"化学意义文本化实现大语言模型对金属有机框架的可解释验证","primary_category":"cond-mat.mtrl-sci","date":"2026-08-13","score":0,"bucket":"other","tags":["材料科学","LLM应用","结构验证"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.11283","has_summary":false},{"id":"2608.11539","title":"Player Perceptions of Generative AI in Games: A Steam Review Analysis","zh_title":"玩家对游戏中生成式AI的感知：Steam评论分析","primary_category":"cs.HC","date":"2026-08-13","score":0,"bucket":"other","tags":["游戏AI","玩家感知","生成式AI"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.11539","has_summary":false},{"id":"2608.12059","title":"Reconfiguring Geovisualization in the Age of Generative AI: Insights from Domain Experts","zh_title":"生成式人工智能时代地理可视化的重构：来自领域专家的见解","primary_category":"cs.CY","date":"2026-08-13","score":0,"bucket":"other","tags":["生成式人工智能","地理可视化","专家访谈"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.12059","has_summary":false},{"id":"2608.11371","title":"Do People Follow AI Advice? Evidence from a Pension Portfolio Choice Experiment","zh_title":"人们会遵循AI建议吗？来自养老金投资组合选择实验的证据","primary_category":"econ.GN","date":"2026-08-13","score":0,"bucket":"other","tags":["AI建议采纳","养老金投资","实验经济学"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.11371","has_summary":false},{"id":"2608.05224","title":"Small Foundation Models of Human Cognition and Behaviour","zh_title":"人类认知与行为的小型基础模型","primary_category":"cs.AI","date":"2026-08-12","score":9,"bucket":"selected","tags":["LLM仿真","认知代理","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.05224","has_summary":true},{"id":"2608.09937","title":"Carefully Considering Culture: Analyzing LLM Alignment in Single- and Multi-Cultural Settings using Cultural Consensus Theory","zh_title":"审慎考量文化：利用文化共识理论分析单文化与多文化环境下大语言模型的对齐","primary_category":"cs.CL","date":"2026-08-12","score":8,"bucket":"selected","tags":["文化仿真","人类数据对照","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.09937","has_summary":true},{"id":"2608.10186","title":"The Deliberative Deficit: An Empirical Critique of LLMs in Democratic Discourse","zh_title":"协商赤字：对民主话语中LLM的实证批判","primary_category":"cs.MA","date":"2026-08-12","score":8,"bucket":"selected","tags":["LLM仿真","民主协商","人类数据对照"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.10186","has_summary":true},{"id":"2608.09790","title":"CARD: Controlled Agentic Reddit Discussions for Credit Card Simulation","zh_title":"CARD：用于信用卡模拟的可控代理式Reddit讨论","primary_category":"cs.AI","date":"2026-08-12","score":5,"bucket":"other","tags":["社会模拟","文本生成","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.09790","has_summary":false},{"id":"2608.07505","title":"Position: We Need Large Language Models Optimized For Our Well-Being","zh_title":"立场：我们需要为人类福祉优化的大语言模型","primary_category":"cs.CY","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM福祉","人机交互","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.07505","has_summary":false},{"id":"2608.10154","title":"Multimodal Item Parameter Estimation using Simulated Response Probabilitie","zh_title":"使用模拟响应概率的多模态项目参数估计","primary_category":"cs.CL","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM仿真","项目反应理论","合成数据"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.10154","has_summary":false},{"id":"2608.10475","title":"Evaluating Rational Contracting in Natural Language","zh_title":"评估自然语言中的理性合同行为","primary_category":"cs.AI","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM agent","合同谈判","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.10475","has_summary":false},{"id":"2608.10703","title":"Your LLM, Your Style: Behavioral Mode Axes for LLM Behavioral Control","zh_title":"你的LLM，你的风格：用于LLM行为控制的行为模式轴","primary_category":"cs.LG","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM人格测量","行为控制","心理测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.10703","has_summary":false},{"id":"2608.09946","title":"HoosierHelp: Benchmarking LLM Agents for Social Service Navigation","zh_title":"HoosierHelp：面向社会服务导航的LLM智能体基准测试","primary_category":"cs.HC","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.09946","has_summary":false},{"id":"2608.10412","title":"When the Interviewer Is a Bot: Behavior, Breakdowns, and Trust in MLLM-Led Interviews","zh_title":"当访谈者是机器人：MLLM主导访谈中的行为、故障与信任","primary_category":"cs.HC","date":"2026-08-12","score":5,"bucket":"other","tags":["MLLM访谈","人机交互","定性研究自动化"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.10412","has_summary":false},{"id":"2608.10262","title":"Not a Monolith: Lab-Level Divergence in the Cooperative Equilibria of Chinese Frontier LLM Agents","zh_title":"并非铁板一块：中国前沿LLM智能体合作均衡的实验室层面差异","primary_category":"cs.MA","date":"2026-08-12","score":5,"bucket":"other","tags":["LLM智能体","囚徒困境","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.10262","has_summary":false},{"id":"2608.10042","title":"UserToolBench: A User-Profile-Hidden Benchmark for Personalized Decision Making in Tool-Use LLMs","zh_title":"UserToolBench：面向工具使用LLM的个性化决策的用户画像隐藏基准","primary_category":"cs.LG","date":"2026-08-12","score":5,"bucket":"other","tags":["个性化决策","工具使用LLM","基准测试"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.10042","has_summary":false},{"id":"2608.08775","title":"OmnilingualGAIA2: Evaluating the Multilingual Gap in Frontier AI Agents","zh_title":"OmnilingualGAIA2：评估前沿AI代理的多语言差距","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["AI代理评测","多语言基准","工具编排"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08775","has_summary":false},{"id":"2608.08868","title":"Conversation as Measurement in Clinical Encounters: Observable Phase Structure, Partially Observable Patient State","zh_title":"临床对话中的测量：可观察的阶段结构与部分可观察的患者状态","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["对话分析","临床NLP","可观测性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08868","has_summary":false},{"id":"2608.08605","title":"ForestBench: A Unified Graph Framework for Evaluating Multi-Agent Collaboration","zh_title":"ForestBench：评估多智能体协作的统一图框架","primary_category":"cs.AI","date":"2026-08-12","score":0,"bucket":"other","tags":["多智能体系统","评估框架","协作图"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08605","has_summary":false},{"id":"2608.09248","title":"Emotion2Skill: Model-Internal Emotion Signals for Adaptive Skill Selection and Evolution","zh_title":"Emotion2Skill：利用模型内部情绪信号实现自适应技能选择与进化","primary_category":"cs.AI","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM智能体","技能路由","内部表征"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09248","has_summary":false},{"id":"2608.08422","title":"Population-Level Generative Modeling for Ranking Data","zh_title":"排序数据的群体级生成建模","primary_category":"stat.ME","date":"2026-08-12","score":0,"bucket":"other","tags":["生成模型","排序数据","统计建模"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08422","has_summary":false},{"id":"2608.09936","title":"Conflict or Strategy? Asymmetric Role Framing of La France insoumise and Rassemblement National in French News Headlines, 2022-2025","zh_title":"冲突还是策略？2022-2025年法国新闻标题中不屈法国与国民联盟的不对称角色框架","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["框架分析","计算传播学","LLM标注"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09936","has_summary":false},{"id":"2608.10258","title":"TAF-MED: Multi-Turn Safety Refusal Collapse in LLMs Under Declared Self-Treatment Intent","zh_title":"TAF-MED：声明自我治疗意图下LLM的多轮安全拒绝崩溃","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM安全","医疗对话","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10258","has_summary":false},{"id":"2608.10299","title":"Co-Evolution in Agentic Systems: Toward Self-Directed Evolution Beyond Human Design","zh_title":"智能体系统中的共同演化：迈向超越人类设计的自我导向演化","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["多智能体系统","共同演化","自演化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10299","has_summary":false},{"id":"2608.10315","title":"Is This Your Final Answer? Cross-Contextual Consistency as a Measure of LLM Credibility","zh_title":"这是你的最终答案吗？跨上下文一致性作为LLM可信度的度量","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM可信度","跨上下文一致性","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10315","has_summary":false},{"id":"2608.10692","title":"SPIEval: Evaluating Large Language Models as Mobile Assistants over Scattered Personal Information","zh_title":"SPIEval：评估大语言模型作为移动助手处理分散个人信息的能力","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM评测","移动助手","个人信息处理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10692","has_summary":false},{"id":"2608.11200","title":"ConVAWG: A Retrieval-Grounded Framework for Controlled Synthetic Dialogue Generation in Violence Against Women and Girls","zh_title":"ConVAWG：面向暴力侵害妇女和女童行为的检索增强可控合成对话生成框架","primary_category":"cs.CL","date":"2026-08-12","score":0,"bucket":"other","tags":["合成对话生成","角色扮演","暴力侵害妇女"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.11200","has_summary":false},{"id":"2608.09988","title":"OpenPM: Auditable Point-in-Time Evaluation for LLM Portfolio-Management Agents","zh_title":"OpenPM：面向LLM投资组合管理代理的可审计时点评估框架","primary_category":"cs.CE","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM代理","金融交易","评估框架"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09988","has_summary":false},{"id":"2608.10126","title":"Procedural Fairness Failures in RLHF from Preference Averaging","zh_title":"RLHF中因偏好平均导致的程序公平性失效","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["RLHF","公平性","偏好聚合"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10126","has_summary":false},{"id":"2608.10218","title":"Mind Viruses: Self-Propagating Ideas in Multi-Agent LLM Systems","zh_title":"思维病毒：多智能体LLM系统中的自我传播思想","primary_category":"cs.AI","date":"2026-08-12","score":0,"bucket":"other","tags":["多智能体系统","AI安全","思想传播"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10218","has_summary":false},{"id":"2608.10329","title":"Who Gets Heeded? An Obligation-Level Audit of Responsiveness in EPA Rulemaking","zh_title":"谁被听取？EPA规则制定中响应度的义务层面审计","primary_category":"cs.CY","date":"2026-08-12","score":0,"bucket":"other","tags":["规则制定","公众评论","AI辅助审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10329","has_summary":false},{"id":"2608.10672","title":"Longitudinal Evidence That General-Purpose Chatbots Actively Foster Relational Engagement","zh_title":"通用聊天机器人主动促进关系性参与的纵向证据","primary_category":"cs.HC","date":"2026-08-12","score":0,"bucket":"other","tags":["人机互动","情感纽带","聊天机器人"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.10672","has_summary":false},{"id":"2608.11027","title":"Mapping and Measuring the Behavioral Evolution of Large Language Models","zh_title":"映射与测量大语言模型的行为演化","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["模型行为分析","嵌入空间比较","无人类对照"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11027","has_summary":false},{"id":"2608.11197","title":"Beyond a Bag of Features: Set-Level Instability in Sparse Autoencoders","zh_title":"超越特征袋：稀疏自编码器中的集合级不稳定性","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["稀疏自编码器","模型可解释性","概念表征"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11197","has_summary":false},{"id":"2608.10818","title":"AI-Generated Interactive Fiction for Educational Use: A Pilot Study of Perceived Comprehensibility, Coherence, and Engagement","zh_title":"用于教育的AI生成互动小说：一项关于感知可理解性、连贯性和参与度的初步研究","primary_category":"cs.HC","date":"2026-08-12","score":0,"bucket":"other","tags":["互动小说","教育技术","用户体验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.10818","has_summary":false},{"id":"2608.11090","title":"Who Uses Open-Weight Models? China and the Shifting Geography of AI in Science","zh_title":"谁在使用开放权重模型？中国与科学中AI地理格局的变迁","primary_category":"cs.CY","date":"2026-08-12","score":0,"bucket":"other","tags":["科学计量学","模型选择","开放科学"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.11090","has_summary":false},{"id":"2608.10046","title":"Detecting Soft Skills in ML Engineering Roles CVs","zh_title":"检测机器学习工程角色简历中的软技能","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["NLP信息抽取","简历分析","软技能检测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10046","has_summary":false},{"id":"2608.10089","title":"Status Association Does Not Reliably Predict Decision Leakage","zh_title":"地位关联不能可靠预测决策泄漏","primary_category":"stat.AP","date":"2026-08-12","score":0,"bucket":"other","tags":["模型偏见","决策泄漏","公平性评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10089","has_summary":false},{"id":"2608.10268","title":"Toward Human Rights Benchmarking for LLMs: A Pilot Methodology","zh_title":"面向LLM的人权基准测试：一项试点方法","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["LLM评测","人权法","法律推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.10268","has_summary":false},{"id":"2608.10175","title":"Beyond Cash Flows: A Multi-Agent AI Framework for Valuing Clinical-Stage, Cross-Border Biotechnology","zh_title":"超越现金流：评估临床阶段跨境生物技术的多智能体AI框架","primary_category":"cs.MA","date":"2026-08-12","score":0,"bucket":"other","tags":["多智能体系统","投资分析","生物技术估值"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10175","has_summary":false},{"id":"2608.10050","title":"Observational Policy Ranking for SMB Financial Guidance from Multi-Action Accounting Logs","zh_title":"基于多动作会计日志的中小企业财务指导观测性策略排序","primary_category":"cs.LG","date":"2026-08-12","score":0,"bucket":"other","tags":["策略学习","会计日志","中小企业"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10050","has_summary":false},{"id":"2608.10532","title":"Benchmarking LLM-Guided Control-Plane Policies for Backend Fault Isolation in HAProxy","zh_title":"基于LLM引导的控制平面策略在HAProxy后端故障隔离中的基准测试","primary_category":"cs.NI","date":"2026-08-12","score":0,"bucket":"other","tags":["负载均衡","故障隔离","LLM控制策略"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.10532","has_summary":false},{"id":"2608.07498","title":"Knowing You Is Everything: LLM Agents Achieve Near-Perfect Profile-Consistent Reaction Prediction in Social Media Simulation","zh_title":"知你即一切：LLM代理在社交媒体模拟中实现近乎完美的画像一致性反应预测","primary_category":"cs.HC","date":"2026-08-11","score":10,"bucket":"selected","tags":["LLM人类仿真","社交媒体模拟","算法保真度"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.07498","has_summary":true},{"id":"2608.09717","title":"How Do Large Language Models Judge Social Attraction? Evidence from Theory-Grounded Persona Ratings Across Multiple LLMs and Humans","zh_title":"大语言模型如何判断社交吸引力？基于理论驱动的人物画像在多个LLM和人类中的评分证据","primary_category":"cs.CL","date":"2026-08-11","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","社交判断偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.09717","has_summary":true},{"id":"2608.08691","title":"EnergyBridge: Benchmarking Household Energy Management, User Participation, and Grid Flexibility","zh_title":"EnergyBridge：家庭能源管理、用户参与和电网灵活性的基准测试","primary_category":"cs.AI","date":"2026-08-11","score":9,"bucket":"selected","tags":["LLM人类仿真","用户参与模拟","电网灵活性"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.08691","has_summary":true},{"id":"2608.07490","title":"Experience-Sensitive Game Learning: A Behavioral Study of Humans and Language Agents","zh_title":"经验敏感的游戏学习：人类与语言代理的行为研究","primary_category":"cs.HC","date":"2026-08-11","score":9,"bucket":"selected","tags":["LLM人类仿真","行为博弈","算法保真度"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.07490","has_summary":true},{"id":"2608.08227","title":"Focus particles and scalar inferences across humans and language models","zh_title":"焦点粒子与标量推理：人类与语言模型的跨系统比较","primary_category":"cs.CL","date":"2026-08-11","score":7,"bucket":"pending","tags":["人类仿真","标量推理","语言模型对比"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.08227","has_summary":true},{"id":"2608.07497","title":"EvalConvoLearn: An Open-Source Framework for Evaluating Grounded Learner Simulations in Tutoring Conversations","zh_title":"EvalConvoLearn：评估辅导对话中基于真实数据的学习者模拟的开源框架","primary_category":"cs.HC","date":"2026-08-11","score":7,"bucket":"pending","tags":["学习者模拟","对话质量评估","真实数据对照"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2608.07497","has_summary":true},{"id":"2608.07538","title":"When LLM Agents Negotiate: Private Information and Dynamic Bargaining in Supply Chains","zh_title":"当LLM智能体谈判：供应链中的私有信息与动态议价","primary_category":"cs.AI","date":"2026-08-11","score":7,"bucket":"pending","tags":["LLM仿真","经济博弈","算法审计"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.07538","has_summary":true},{"id":"2608.08199","title":"Persuasive and Compliant Tendencies Predict Group Decision-Making in Humans and Language Models","zh_title":"说服与顺从倾向预测人类和语言模型中的群体决策","primary_category":"cs.AI","date":"2026-08-11","score":7,"bucket":"pending","tags":["LLM群体决策","行为倾向测量","人机对照"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.08199","has_summary":true},{"id":"2608.09574","title":"The Politician, the Liar, and the Obedient Worker: Emerging Behavior of LLM Agents in Hierarchical Games","zh_title":"政客、说谎者与顺从的工人：层级博弈中LLM智能体的涌现行为","primary_category":"cs.AI","date":"2026-08-11","score":7,"bucket":"pending","tags":["LLM仿真","行为博弈","多智能体"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.09574","has_summary":true},{"id":"2608.00818","title":"The Scaling Paradox in Human-AI Collaboration","zh_title":"人机协作中的规模悖论","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["人机协作","规模法则","行为建模"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.00818","has_summary":false},{"id":"2608.02971","title":"Mapping the City Through the Lens of Language Models","zh_title":"通过语言模型的镜头绘制城市地图","primary_category":"cs.CL","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM测量","城市认知","隐含假设"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.02971","has_summary":false},{"id":"2608.06123","title":"Poli-Bias: Understanding and Measuring Large Language Model Biases in International Political Conflicts","zh_title":"Poli-Bias：理解和测量国际政治冲突中大语言模型的偏见","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM偏见","政治冲突","反事实框架"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.06123","has_summary":false},{"id":"2608.07641","title":"SurveyReview: A Reviewer-Aligned Benchmark for Survey Evaluators","zh_title":"SurveyReview：面向综述评估者的审稿人对齐基准","primary_category":"cs.CL","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM评估","审稿对齐","基准数据集"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.07641","has_summary":false},{"id":"2608.07812","title":"On the use of foundation models in cognitive science","zh_title":"论基础模型在认知科学中的应用","primary_category":"cs.CL","date":"2026-08-11","score":5,"bucket":"other","tags":["认知建模","行为对齐","方法论框架"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.07812","has_summary":false},{"id":"2608.08942","title":"Same Question, Different Answer? Measuring and Mitigating Prompt Privilege for Equitable AI Access","zh_title":"相同问题，不同答案？测量与缓解提示特权以实现公平的AI访问","primary_category":"cs.CL","date":"2026-08-11","score":5,"bucket":"other","tags":["AI公平性","提示工程","可访问性"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.08942","has_summary":false},{"id":"2608.07499","title":"Evaluation of Motivational Interviewing Counsellors with Task-Aware Multi-Stage LLM-Based Simulated Clients","zh_title":"基于任务感知多阶段LLM模拟来访者的动机性访谈咨询师评估","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM模拟来访者","动机性访谈","评估框架"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.07499","has_summary":false},{"id":"2608.08881","title":"Theory-Guided Deception Detection: A RAG-Based Artificial Intelligence Exploration","zh_title":"理论引导的欺骗检测：基于RAG的人工智能探索","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["欺骗检测","RAG","LLM标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.08881","has_summary":false},{"id":"2608.07762","title":"Who Verifies the Benchmark? Decentralizing Trust in Large Language Model Evaluation","zh_title":"谁验证基准？大语言模型评估中的信任去中心化","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM评估","去中心化验证","评估偏差"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.07762","has_summary":false},{"id":"2608.08026","title":"The Authority Expectancy Effect in Multi-User Conflict","zh_title":"多用户冲突中的权威期望效应","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM社会模拟","权威效应","冲突调解"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.08026","has_summary":false},{"id":"2608.08061","title":"CORDA: A Benchmark for Hierarchical Harm-Centric Moral Reasoning in Large Language Models","zh_title":"CORDA：面向大语言模型的以伤害为中心的层级化道德推理基准","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["道德推理","LLM评估","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.08061","has_summary":false},{"id":"2608.08621","title":"Business Arena: Benchmarking LLM Agents in a Realistic Marketplace","zh_title":"商业竞技场：在真实市场环境中评测大语言模型智能体","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM Agent","市场模拟","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.08621","has_summary":false},{"id":"2608.09485","title":"Capability Is Not Propensity: Measuring Pressure-Robust Cooperative Behavior in Civic LLM Agents","zh_title":"能力非倾向：测量公民LLM智能体的抗压合作行为","primary_category":"cs.AI","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM合作行为","压力测试","社会推理"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.09485","has_summary":false},{"id":"2608.07481","title":"Cross-Model Humor Preference Modeling with Cards Against Humanity","zh_title":"基于Cards Against Humanity的跨模型幽默偏好建模","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM偏好建模","幽默选择","跨模型对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.07481","has_summary":false},{"id":"2608.07512","title":"EMMR: Emotion-Mediated Multimodal Reasoning for Personality Assessment in Asynchronous Video Interviews","zh_title":"EMMR：异步视频面试中基于情绪中介的多模态推理人格评估","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["人格评估","多模态推理","异步视频面试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.07512","has_summary":false},{"id":"2608.07523","title":"From Evaluated Models to Evaluation Aids: A Multi-Evidence Study of LLM-Based Difficulty Calibration for Programming Examinations","zh_title":"从被评估模型到评估辅助工具：基于多证据的编程考试难度校准研究","primary_category":"cs.CY","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM评估辅助","试题难度校准","编程教育"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.07523","has_summary":false},{"id":"2608.07488","title":"Large Language Models Explain Experts Better Than Experts Themselves","zh_title":"大语言模型比专家自己更能解释专家","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["隐性知识外化","LLM 知识提取","决策支持"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.07488","has_summary":false},{"id":"2608.07496","title":"Human-Simulation Interaction: From Prediction to Exploration in LLM Agent Simulations for Policy","zh_title":"人-仿真交互：从预测到探索的LLM智能体政策模拟","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","政策评估"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.07496","has_summary":false},{"id":"2608.08497","title":"SocialFiVis: A Visual Analytics Sandbox for LLM-Grounded Multi-Agent Simulation in Social Finance","zh_title":"SocialFiVis：面向社交金融中基于LLM的多智能体仿真的可视分析沙盒","primary_category":"cs.HC","date":"2026-08-11","score":5,"bucket":"other","tags":["多智能体仿真","社交金融","可视分析"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.08497","has_summary":false},{"id":"2608.07511","title":"How sensitive do we want AI to be? Socio-communicative competencies of large language models in healthcare","zh_title":"我们希望AI有多敏感？医疗保健中大语言模型的社会沟通能力","primary_category":"cs.HC","date":"2026-08-11","score":3,"bucket":"other","tags":["LLM社交能力","医疗对话","角色扮演评测"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07511","has_summary":false},{"id":"2608.07495","title":"EmoPatient: An Emotion-Directed Patient Simulator for Realistic Palliative Care Communication Training","zh_title":"EmoPatient：面向情感引导的姑息治疗沟通训练患者模拟器","primary_category":"cs.HC","date":"2026-08-11","score":3,"bucket":"other","tags":["患者模拟","沟通训练","情感动态"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07495","has_summary":false},{"id":"2607.23621","title":"GEMCo: A Validated, Ethically Releasable Proxy for Inaccessible Counselling Data","zh_title":"GEMCo：一个经过验证、可伦理发布的不可访问咨询数据代理","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["数据代理","咨询对话","隐私保护"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.23621","has_summary":false},{"id":"2607.23065","title":"Touching or Chatting: The Utility of LLMs and Tactile Charts for Learning about Complex Chart Types by BLV Individuals","zh_title":"触摸还是聊天：大语言模型与触觉图表对盲人及低视力者学习复杂图表类型的效用","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["辅助技术","人机交互","盲人教育"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.23065","has_summary":false},{"id":"2607.23705","title":"Social learning drives underprioritization of collective challenges","zh_title":"社会学习导致集体挑战的低优先级","primary_category":"physics.soc-ph","date":"2026-08-11","score":0,"bucket":"other","tags":["社会学习","集体行动","动态模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23705","has_summary":false},{"id":"2607.26977","title":"TREK: A Travel Reasoning and Evaluation Kit for LLM Agents in Complex Trip Planning","zh_title":"TREK：面向复杂旅行规划的LLM智能体旅行推理与评估套件","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM智能体","旅行规划","基准评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26977","has_summary":false},{"id":"2607.28650","title":"Unanticipated Effects of Generative AI on Expertise Pathways and Performance Perception in System Administration","zh_title":"生成式AI对系统管理专业知识路径与绩效感知的意外影响","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["人机交互","质性研究","专业知识发展"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28650","has_summary":false},{"id":"2608.01176","title":"When Words Divide: Diachronic Ideological Polarization in Political Discourse on Social Media","zh_title":"当词语分裂：社交媒体政治话语中的历时意识形态极化","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["意识形态极化","社交媒体分析","词嵌入"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01176","has_summary":false},{"id":"2608.02311","title":"AI Governance for Institutional Readiness in Finance","zh_title":"面向金融领域机构就绪度的人工智能治理","primary_category":"econ.EM","date":"2026-08-11","score":0,"bucket":"other","tags":["AI治理","金融风险管理","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02311","has_summary":false},{"id":"2608.05656","title":"Studying People to Study AI: Expert Perspectives on the Epistemic Fit and Barriers of Human Research in AI Safety & Ethics","zh_title":"通过研究人来研究AI：专家对AI安全与伦理中人类研究方法的认知契合与障碍的看法","primary_category":"cs.CY","date":"2026-08-11","score":0,"bucket":"other","tags":["AI安全","专家调查","方法论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05656","has_summary":false},{"id":"2608.06609","title":"Automated item evaluation: Predicting item acceptance and rejection using LLM-generated critiques","zh_title":"自动化试题评估：利用LLM生成的评语预测试题接受与拒绝","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["自动化试题评估","NLP分类","教育测量"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06609","has_summary":false},{"id":"2608.08160","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","zh_title":"LLM智能体能按剧本走吗？交互叙事中长期一致性基准测试","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["交互叙事","一致性评测","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.08160","has_summary":false},{"id":"2608.08164","title":"STEMMA: An Adversarial Multi-Agent Framework for Evaluating Self-Identity Consistency in LLMs","zh_title":"STEMMA：用于评估大语言模型自我身份一致性的对抗性多智能体框架","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体系统","模型身份一致性","对抗性提示"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08164","has_summary":false},{"id":"2608.08606","title":"Mitigating Gender Bias in English to Romanian Machine Translation","zh_title":"缓解英罗机器翻译中的性别偏见","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["机器翻译","性别偏见","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08606","has_summary":false},{"id":"2608.08975","title":"How Can Rhetoric Reward-Hack AI Reviewers? Dissecting Rhetorical Sensitivity in AI-Based Peer Review","zh_title":"修辞如何奖励黑客AI审稿人？剖析AI同行评审中的修辞敏感性","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["AI审稿","修辞敏感性","奖励黑客"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08975","has_summary":false},{"id":"2608.09080","title":"When Confidence Fails: Overconfidence in LLMs under Uncertainty and Missing Clinical Information","zh_title":"当置信度失效：不确定性及缺失临床信息下大语言模型的过度自信","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["置信度校准","医学问答","模型可靠性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09080","has_summary":false},{"id":"2608.09128","title":"Social Gym and SPaRTan: Benchmarking and Improving LLM Social Reasoning via Multi-Agent Game Tournaments","zh_title":"Social Gym与SPaRTan：通过多智能体游戏锦标赛基准测试与提升大语言模型社交推理能力","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体","社交推理","游戏基准"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09128","has_summary":false},{"id":"2608.09189","title":"EmoS: A Theory-Grounded Framework for Evaluating and Aligning Emotional Intelligence in Spoken Language Models","zh_title":"EmoS：一个基于理论的框架，用于评估和对齐口语语言模型中的情商","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["情感智能","口语语言模型","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.09189","has_summary":false},{"id":"2608.09420","title":"Intent Speaks Louder: Controllable User Simulation Beyond Response Imitation","zh_title":"意图胜于言辞：超越响应模仿的可控用户仿真","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["用户仿真","对话系统","意图控制"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.09420","has_summary":false},{"id":"2608.09510","title":"Build it, Break it, Repeat: Benchmarking and improving LLM-manipulated disinformation detection in social media posts","zh_title":"构建、破坏、重复：基准测试与改进社交媒体帖子中LLM操纵的虚假信息检测","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["虚假信息检测","对抗攻击","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09510","has_summary":false},{"id":"2608.09925","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","zh_title":"从价值观到基准：评估荷兰政府使用的大语言模型","primary_category":"cs.CL","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM评测","政府应用","基准测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09925","has_summary":false},{"id":"2608.07517","title":"The Judge Knows When It Knows: Calibrated Abstention for LLM-Based A/B-Test Prediction","zh_title":"法官知道何时自知：基于LLM的A/B测试预测的校准弃权","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["A/B测试","LLM评估","校准弃权"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07517","has_summary":false},{"id":"2608.07537","title":"An evolutionary model of animats with VLM-based subjective evaluation","zh_title":"基于VLM主观评价的animat进化模型","primary_category":"cs.NE","date":"2026-08-11","score":0,"bucket":"other","tags":["进化计算","虚拟机器人","视觉语言模型"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.07537","has_summary":false},{"id":"2608.07614","title":"DevIntent: How Much Does LLM-Generated Code Violate Developer Intent?","zh_title":"DevIntent：LLM生成的代码在多大程度上违背开发者意图？","primary_category":"cs.SE","date":"2026-08-11","score":0,"bucket":"other","tags":["代码生成","意图违背","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07614","has_summary":false},{"id":"2608.07688","title":"IntelliAudit: Using Large Language Models to Evaluate Audit Controls","zh_title":"IntelliAudit：使用大语言模型评估审计控制","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体系统","审计自动化","决策支持"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07688","has_summary":false},{"id":"2608.08126","title":"Accurate Ensembles, Fragile Narratives: Multi-Scale Stacking and a Fidelity Audit of LLM-Generated Explanations for Credit Risk","zh_title":"准确集成，脆弱叙事：多尺度堆叠与LLM生成信用风险解释的保真度审计","primary_category":"cs.LG","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM解释生成","信用评分","保真度审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08126","has_summary":false},{"id":"2608.08143","title":"DS@GT ARC at Touch\\'e: Large Language Models for Retrieval-Augmented Debate","zh_title":"DS@GT ARC在Touché 2025检索增强辩论任务中的大语言模型应用","primary_category":"cs.IR","date":"2026-08-11","score":0,"bucket":"other","tags":["辩论生成","LLM评估","检索增强"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.08143","has_summary":false},{"id":"2608.08822","title":"Automated Generation of Complexity-Validated Decision Scenarios Using Large Language Models","zh_title":"使用大语言模型自动生成复杂度验证的决策场景","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM生成","决策场景","复杂度验证"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08822","has_summary":false},{"id":"2608.08885","title":"Towards an LLM-based method for quantifying the sexual content in song lyrics","zh_title":"基于大语言模型的量化歌曲歌词性内容的方法","primary_category":"physics.soc-ph","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM标注","歌词分析","内容量化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08885","has_summary":false},{"id":"2608.09019","title":"How People Evaluate AI-, Expert-, and Peer-Style Financial Advice","zh_title":"人们如何评估AI、专家和同伴风格的财务建议","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["人机交互","财务建议","来源归因"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.09019","has_summary":false},{"id":"2608.09282","title":"ComboShoppingBench: Evaluating LLM Agents for Budget-Constrained Basket Shopping with Coupons","zh_title":"ComboShoppingBench：评估预算约束下使用优惠券的组合购物LLM智能体","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM智能体","购物基准","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09282","has_summary":false},{"id":"2608.09638","title":"Avalon-ToM-Bench: Evaluating Fine-Grained Theory of Mind via Asymmetric Game Mechanics","zh_title":"Avalon-ToM-Bench：通过非对称游戏机制评估细粒度心理理论","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["心理理论","多智能体","基准评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09638","has_summary":false},{"id":"2608.09861","title":"Towards Expert-level Medical AI for Real-time Video Consultations","zh_title":"面向实时视频会诊的专家级医学人工智能","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["医学AI","多智能体系统","视频问诊"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09861","has_summary":false},{"id":"2608.07642","title":"Contextual Value Alignment via Multilayer Combinatorial Fusion","zh_title":"基于多层组合融合的上下文价值对齐","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["价值对齐","多智能体","组合融合"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07642","has_summary":false},{"id":"2608.08159","title":"When Is a Steerable Concept Representation Real? Measurement Confounds in a Cross-Family Audit of Neuroscience Parallels in LLMs","zh_title":"可操控概念表征何时为真？跨模型家族审计LLM中神经科学类比的测量混淆","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["AI神经科学","线性探测","激活操控"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08159","has_summary":false},{"id":"2608.08240","title":"A Fair Objective for Human-Empowerment-Preserving AI: Desiderata, Design, and Likely Behavioral Consequences","zh_title":"一种维护人类赋权的公平AI目标：需求、设计与可能的行为后果","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["AI安全","多智能体系统","目标设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08240","has_summary":false},{"id":"2608.08254","title":"Your Prompt Is Not the Only Prompt: How Much Do LLMs Weight Structured-Output Schema Descriptions?","zh_title":"你的提示不是唯一的提示：LLM对结构化输出模式描述的权重有多大？","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["结构化输出","提示工程","模型行为分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08254","has_summary":false},{"id":"2608.08281","title":"Exploring LLM Capabilities for Situational Understanding and COLREG compliance on real-world maritime navigation scenarios","zh_title":"探索大语言模型在真实航海导航场景中的态势理解与COLREG合规能力","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM","航海导航","自动驾驶"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.08281","has_summary":false},{"id":"2608.08284","title":"Fair on the Surface? Benchmarking Hidden-Output Fairness Gaps in LLM Recommenders","zh_title":"表面公平？基准测试LLM推荐系统中的隐藏输出公平性差距","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["推荐系统","公平性审计","表示偏移"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08284","has_summary":false},{"id":"2608.08746","title":"Scale-to-Dialogue: Low-Burden Elicitation of Daily Premenstrual Symptom Ratings with Small Language Models","zh_title":"规模转对话：用小语言模型低负担获取每日经前症状评分","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["症状追踪","对话系统","标签恢复"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08746","has_summary":false},{"id":"2608.08852","title":"Findings of the First Teaching Monster Challenge: A Benchmark of Pedagogical Content Knowledge in AI Agents","zh_title":"首届教学怪物挑战赛发现：AI智能体中教学内容知识的基准测试","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["AI教学","基准测试","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08852","has_summary":false},{"id":"2608.08889","title":"LLM Reasoning for Subjective Tasks: Failure Modes, Mitigation, and Dynamic Reasoning Routing","zh_title":"主观任务中的LLM推理：失败模式、缓解策略与动态推理路由","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM推理","主观验证","偏好对齐"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08889","has_summary":false},{"id":"2608.09253","title":"SkillSentry: Reliable Skill Execution for LLM Agents via Runtime Assurance","zh_title":"SkillSentry：通过运行时保障实现LLM代理的可靠技能执行","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM代理","技能执行","运行时保障"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09253","has_summary":false},{"id":"2608.09343","title":"LLM-Guided Heuristic Design from Simulation Traces: A Case Study in Dynamic Production and AGV Scheduling","zh_title":"基于仿真轨迹的LLM引导启发式设计：动态生产与AGV调度案例研究","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["LLM引导优化","生产调度","仿真优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09343","has_summary":false},{"id":"2608.09848","title":"CEAA: A Cognitive Embodied Agents Architecture for Interactive Computing Systems","zh_title":"面向交互式计算系统的认知具身智能体架构","primary_category":"cs.AI","date":"2026-08-11","score":0,"bucket":"other","tags":["具身智能体","虚拟环境","认知架构"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.09848","has_summary":false},{"id":"2608.07543","title":"Performance of large language models in the optical diagnosis of colorectal polyps","zh_title":"大语言模型在结直肠息肉光学诊断中的性能","primary_category":"cs.CV","date":"2026-08-11","score":0,"bucket":"other","tags":["医学图像诊断","LLM性能评测","内窥镜"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07543","has_summary":false},{"id":"2608.07593","title":"Weather- and Location-Aware Agentic Dining Recommendation: Leveraging LLM World Knowledge for Region-Sensitive Contextual Reasoning","zh_title":"天气与位置感知的智能体餐饮推荐：利用大语言模型世界知识进行区域敏感的情境推理","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["推荐系统","大语言模型","情境感知"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07593","has_summary":false},{"id":"2608.07902","title":"Beyond \"I Can't Help With That\": How Child Safety Experts Evaluate AI Chatbot Safety","zh_title":"超越“我无法帮助”：儿童安全专家如何评估AI聊天机器人安全性","primary_category":"cs.CY","date":"2026-08-11","score":0,"bucket":"other","tags":["AI安全","儿童保护","聊天机器人"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07902","has_summary":false},{"id":"2608.08266","title":"On the Robustness of LLMs' Internal Representation of Code Correctness","zh_title":"大语言模型代码正确性内部表征的鲁棒性研究","primary_category":"cs.SE","date":"2026-08-11","score":0,"bucket":"other","tags":["代码正确性","模型内部表征","鲁棒性分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.08266","has_summary":false},{"id":"2608.08344","title":"PRISM: A Predictive Protocol for Permutation Optimization via Landscape Diagnostics","zh_title":"PRISM：一种基于景观诊断的排列优化预测协议","primary_category":"cs.LG","date":"2026-08-11","score":0,"bucket":"other","tags":["排列优化","景观诊断","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08344","has_summary":false},{"id":"2608.09351","title":"Test-Time Augmentation for LLMs: When Input Diversity Beats Output Diversity at Matched Compute","zh_title":"大语言模型的测试时增强：当输入多样性在匹配计算量下胜过输出多样性","primary_category":"cs.LG","date":"2026-08-11","score":0,"bucket":"other","tags":["测试时增强","推理效率","准确率优化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.09351","has_summary":false},{"id":"2608.09828","title":"Multi-Agent AI Safety as an Institutional Design Problem","zh_title":"作为制度设计问题的多智能体AI安全","primary_category":"cs.LG","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体系统","AI安全","制度设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.09828","has_summary":false},{"id":"2608.09294","title":"Graphing the Everyday: A Neurosymbolic Approach to Eliciting Routines for Just-In-Time Adaptive Interventions","zh_title":"绘制日常生活：一种用于即时自适应干预的神经符号方法","primary_category":"cs.HC","date":"2026-08-11","score":0,"bucket":"other","tags":["对话代理","知识图谱","健康干预"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.09294","has_summary":false},{"id":"2608.07810","title":"Mobility, Memory, and Network Structure in Agent-Based Models of Convention Tipping and Convergence","zh_title":"基于智能体的约定转变与收敛模型中的移动性、记忆和网络结构","primary_category":"cs.MA","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体仿真","临界点动力学","社会约定"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07810","has_summary":false},{"id":"2608.08437","title":"AI and the Research Team","zh_title":"人工智能与研究团队","primary_category":"econ.GN","date":"2026-08-11","score":0,"bucket":"other","tags":["多智能体系统","团队规模","AI自动化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.08437","has_summary":false},{"id":"2608.07367","title":"People Are Not Just Their Countries. Disentangling Social Determinants of LLM Value Alignment Across Europe","zh_title":"人不仅是其国家：解构欧洲LLM价值观对齐的社会决定因素","primary_category":"cs.AI","date":"2026-08-10","score":9,"bucket":"selected","tags":["LLM价值观对齐","社会调查复现","偏差分析"],"rubric_hits":["A1","A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2608.07367","has_summary":true},{"id":"2608.06379","title":"Preventive Care Recommendations by Large Language Models","zh_title":"大语言模型的预防保健建议","primary_category":"cs.HC","date":"2026-08-10","score":9,"bucket":"selected","tags":["LLM仿真","医生决策","人类数据对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.06379","has_summary":true},{"id":"2608.04009","title":"SocietyBench: Forecasting Counterfactual Social-World Evolution","zh_title":"SocietyBench：预测反事实社会世界演化","primary_category":"cs.CL","date":"2026-08-10","score":8,"bucket":"selected","tags":["社会模拟","LLM预测","反事实推理"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.04009","has_summary":true},{"id":"2608.07316","title":"Natural Language Processing Psychometrics","zh_title":"自然语言处理心理测量学","primary_category":"cs.CL","date":"2026-08-10","score":8,"bucket":"selected","tags":["LLM仿真","心理测量","可解释AI"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.07316","has_summary":true},{"id":"2608.06485","title":"Do AI Personas Grow? Analyzing and Benchmarking Personality Evolution in LLM Agents After Life Events","zh_title":"AI人格会成长吗？分析并基准测试LLM智能体经历生活事件后的人格演变","primary_category":"cs.CL","date":"2026-08-10","score":6,"bucket":"other","tags":["LLM人格演变","人类数据对照","仿真可靠性"],"rubric_hits":["D2","A2"],"abs_url":"https://arxiv.org/abs/2608.06485","has_summary":false},{"id":"2608.06977","title":"Confirming Our Biases? Evaluating the Capabilities, Risks, and Societal Impact of Large Language Models","zh_title":"确认我们的偏见？评估大语言模型的能力、风险和社会影响","primary_category":"cs.CL","date":"2026-08-10","score":5,"bucket":"other","tags":["LLM偏见","提示敏感性","模型行为分析"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.06977","has_summary":false},{"id":"2608.07243","title":"Recipes for Creativity: Iterative Generation and Evaluation in Large Language Models","zh_title":"创造力配方：大语言模型中的迭代生成与评估","primary_category":"cs.AI","date":"2026-08-10","score":5,"bucket":"other","tags":["LLM创造力","迭代生成","TTCT评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.07243","has_summary":false},{"id":"2608.06922","title":"Deal Me Maybe: The Role of Emotions in Multi-Agent Negotiation","zh_title":"也许成交：情绪在多智能体谈判中的作用","primary_category":"cs.AI","date":"2026-08-10","score":5,"bucket":"other","tags":["LLM谈判","情绪影响","多智能体模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06922","has_summary":false},{"id":"2608.06949","title":"Does Splitting a Triage Decision Across Agents Hide Bias or Help Catch It? A Multi-Agent Simulation Study of LLM-Based Resource Allocation Under Audit Capacity Constraints","zh_title":"将分诊决策拆分给多个智能体会隐藏偏见还是有助于发现偏见？审计能力约束下基于LLM的资源分配多智能体仿真研究","primary_category":"cs.AI","date":"2026-08-10","score":5,"bucket":"other","tags":["LLM仿真","多智能体","偏见审计"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06949","has_summary":false},{"id":"2608.06955","title":"Critical Acclaim Orientation in Large Language Models: Evidence from Film Preference Elicitation","zh_title":"大语言模型中的好评取向：来自电影偏好诱导的证据","primary_category":"cs.AI","date":"2026-08-10","score":5,"bucket":"other","tags":["LLM偏好","文化偏见","电影评价"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.06955","has_summary":false},{"id":"2608.06980","title":"Social Facilitation of Creative Reflection: AI-agents and Humans","zh_title":"创造性反思的社会促进：AI代理与人类","primary_category":"cs.HC","date":"2026-08-10","score":5,"bucket":"other","tags":["社会模拟","AI代理","创造性反思"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06980","has_summary":false},{"id":"2607.27191","title":"Can AI agents conduct open-ended AI research? Early evidence from two case studies","zh_title":"AI代理能否进行开放式AI研究？来自两个案例的早期证据","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["AI代理","AI研究自动化","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27191","has_summary":false},{"id":"2607.27853","title":"FinanceHarness: Autonomous Financial Deep Research Framework","zh_title":"FinanceHarness：自主金融深度研究框架","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","金融研究","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27853","has_summary":false},{"id":"2608.06539","title":"Don't `Well, Actually' Me Unless You Know What You're Talking About: Weak Presupposition Verification Degrades General QA Performance","zh_title":"别对我说‘其实’除非你懂：弱预设验证会降低通用问答性能","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["虚假预设问答","NLP评测","模型鲁棒性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06539","has_summary":false},{"id":"2608.06589","title":"Beyond \"AI Language\": The case for the idiolectal nature of LLM output","zh_title":"超越“AI语言”：论大语言模型输出的个人方言特性","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["语言风格","文体计量","LLM输出分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06589","has_summary":false},{"id":"2608.06663","title":"The Horizon Gap: Planning, Memory, Execution, Training, and Evaluation for Long-Horizon LLM Agents","zh_title":"视野差距：长时程LLM智能体的规划、记忆、执行、训练与评估","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","长时程任务","智能体评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06663","has_summary":false},{"id":"2608.06718","title":"Do Audio Language Models Use Paralinguistic Evidence? Counterfactual Audits for Response Evaluation","zh_title":"音频语言模型是否使用副语言证据？基于反事实审计的响应评估","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["模型评测","音频语言模型","反事实审计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06718","has_summary":false},{"id":"2608.06785","title":"Multi-Perspective Triad Interaction Graph Neural Network for Cognitive Distortion Detection","zh_title":"用于认知扭曲检测的多视角三元组交互图神经网络","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["认知扭曲检测","图神经网络","心理健康"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06785","has_summary":false},{"id":"2608.07208","title":"Measuring Concept Content in Text from LLM Activations: ESG Evidence from Concept Vectors and Linear Probes","zh_title":"从LLM激活值测量文本概念含量：基于概念向量和线性探针的ESG证据","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["概念测量","线性探针","ESG文本分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07208","has_summary":false},{"id":"2608.07282","title":"Gaze Behavior in Visual World Experiments Can be Modeled With Off-the-shelf Language-Vision Encoders","zh_title":"视觉世界实验中的注视行为可用现成语言-视觉编码器建模","primary_category":"cs.CL","date":"2026-08-10","score":0,"bucket":"other","tags":["计算心理语言学","多模态模型","眼动预测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07282","has_summary":false},{"id":"2608.06735","title":"IB-RL: Isolated Bilateral Reinforcement Learning for Strategic Dialogue Agents","zh_title":"IB-RL：面向策略对话智能体的隔离双边强化学习","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体强化学习","策略对话","博弈训练"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06735","has_summary":false},{"id":"2608.07418","title":"ResidencyRL: Reinforcement Learning in Simulated Clinical Environments","zh_title":"ResidencyRL：模拟临床环境中的强化学习","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["强化学习","临床AI","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07418","has_summary":false},{"id":"2608.06578","title":"Divergent Response Modes in Frontier Language Models Under Steering Pressure","zh_title":"前沿语言模型在引导压力下的分歧响应模式","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["模型行为分析","引导压力","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06578","has_summary":false},{"id":"2608.06632","title":"Shape Your Feed: An LLM-based Agentic System for Conversational Recommendation","zh_title":"塑造你的信息流：基于大语言模型的对话式推荐智能体系统","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["对话推荐","智能体系统","用户偏好对齐"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06632","has_summary":false},{"id":"2608.06871","title":"CEDAR: Agent-Orchestrated Tree Search for Goal-Directed Optimization of Complex Systems","zh_title":"CEDAR：面向复杂系统目标导向优化的智能体编排树搜索","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","复杂系统","树搜索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06871","has_summary":false},{"id":"2608.06926","title":"TRIBE: Predicting Team Performance via Communication Behavior Ensembles","zh_title":"TRIBE：通过通信行为集成预测团队表现","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","团队协作","行为模式"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06926","has_summary":false},{"id":"2608.07077","title":"Transformers Struggle to Use Their Emergent World Models: Revisiting the Tower of Hanoi, and the Illusion of Thinking","zh_title":"Transformer难以使用其涌现的世界模型：重访汉诺塔与思维的幻觉","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["可解释性","规划推理","世界模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07077","has_summary":false},{"id":"2608.07202","title":"Authoring and Management of Transparent Research Integrity Assessments of Randomised Clinical Trial Publications Using LLM-assisted Tools and Provenance Knowledge Graphs","zh_title":"利用LLM辅助工具和溯源知识图谱对随机临床试验出版物进行透明研究诚信评估的创作与管理","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["研究诚信","LLM辅助工具","知识图谱"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07202","has_summary":false},{"id":"2608.07220","title":"Beyond the Black Box: Interpretable Models of Human Randomisation Failures","zh_title":"超越黑箱：人类随机化失败的可解释模型","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["行为博弈","可解释模型","人类决策"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07220","has_summary":false},{"id":"2608.07437","title":"Fisher-R1: Training LLM Agents for Reliable Hypothesis Testing","zh_title":"Fisher-R1：训练LLM智能体进行可靠假设检验","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["LLM智能体","统计推理","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07437","has_summary":false},{"id":"2608.07457","title":"Interaction Creates Dynamical AI Behavior Absent in Isolation","zh_title":"交互创造孤立时不存在的动态AI行为","primary_category":"cs.AI","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体交互","非平衡物理","AI行为动力学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07457","has_summary":false},{"id":"2608.06811","title":"Coupling Planning with Episodic Memory in LLM Agents for Software Issue Resolution","zh_title":"在LLM智能体中耦合规划与情景记忆以解决软件问题","primary_category":"cs.SE","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","软件工程","规划与记忆"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06811","has_summary":false},{"id":"2608.07091","title":"Human-Centered Explainable AI for TinyML Edge Devices: A Pareto-Based Selection Framework with LLM-Guided Design","zh_title":"面向TinyML边缘设备的人本可解释AI：基于帕累托的选择框架与LLM引导设计","primary_category":"cs.HC","date":"2026-08-10","score":0,"bucket":"other","tags":["可解释AI","边缘计算","多目标优化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07091","has_summary":false},{"id":"2608.06804","title":"Fact-Check Your Information (FYI): A Design Probe to Understand How People Actually Fact-Check Data-Driven Articles","zh_title":"事实核查你的信息：一项理解人们如何实际核查数据驱动文章的设计探针","primary_category":"cs.HC","date":"2026-08-10","score":0,"bucket":"other","tags":["事实核查","人机交互","数据新闻"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06804","has_summary":false},{"id":"2608.07093","title":"UncertaintyVis: Preserving Linguistic Uncertainty in Automated Text-to-Chart Generation","zh_title":"UncertaintyVis：在自动文本到图表生成中保留语言不确定性","primary_category":"cs.HC","date":"2026-08-10","score":0,"bucket":"other","tags":["文本到图表","不确定性可视化","人机交互"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.07093","has_summary":false},{"id":"2608.06782","title":"Investigating the Presence and Development of Student Instructor Preferences in a Large-Scale CS1 Course","zh_title":"大规模CS1课程中学生对教师偏好的存在与发展研究","primary_category":"cs.CY","date":"2026-08-10","score":0,"bucket":"other","tags":["计算机教育","学生偏好","学习分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06782","has_summary":false},{"id":"2608.07069","title":"Invisible to the Machine: Auditing AI Restaurant, Cafe, and Bar Recommendation Against a Complete Market Census","zh_title":"机器看不见：基于完整市场普查的AI餐饮推荐审计","primary_category":"cs.IR","date":"2026-08-10","score":0,"bucket":"other","tags":["AI审计","推荐系统","偏差分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.07069","has_summary":false},{"id":"2608.07280","title":"Why Study Emergent Behavior When You Can Regulate It? Aligning Multi-Agent Systems with Reward Prediction","zh_title":"为何研究涌现行为？用奖励预测对齐多智能体系统","primary_category":"cs.MA","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体强化学习","社会困境","涌现行为调控"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07280","has_summary":false},{"id":"2608.07295","title":"Learning Long-Term Educational Investment Policies under Residential Sorting","zh_title":"居住分选下的长期教育投资政策学习","primary_category":"cs.MA","date":"2026-08-10","score":0,"bucket":"other","tags":["多智能体系统","强化学习","教育政策模拟"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.07295","has_summary":false},{"id":"2608.06741","title":"Solver-Guided Reasoning for Mixed-Equilibrium Strategies","zh_title":"求解器引导的混合均衡策略推理","primary_category":"cs.LG","date":"2026-08-10","score":0,"bucket":"other","tags":["博弈论","LLM推理","均衡求解"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06741","has_summary":false},{"id":"2608.06842","title":"Tabular Foundation Models and the Unity of Economic Behaviour","zh_title":"表格基础模型与经济行为的统一性","primary_category":"econ.GN","date":"2026-08-10","score":0,"bucket":"other","tags":["表格基础模型","经济行为预测","随机效用模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06842","has_summary":false},{"id":"2608.06733","title":"Estimating GHG Emissions from AI Use: Framework for Corporate-Level Measurement","zh_title":"估算AI使用的温室气体排放：企业级测量框架","primary_category":"physics.soc-ph","date":"2026-08-10","score":0,"bucket":"other","tags":["碳排放核算","AI环境影响","企业标准"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06733","has_summary":false},{"id":"2608.06771","title":"Agentic Artificial Intelligence for Reproducible Human-in-the-Loop Environmental Health Research","zh_title":"面向可重复的人机协同环境健康研究的智能体人工智能","primary_category":"physics.soc-ph","date":"2026-08-10","score":0,"bucket":"other","tags":["智能体AI","环境健康","人机协同"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06771","has_summary":false},{"id":"2603.00059","title":"Stochastic Parrots or Singing in Harmony? Testing Five Leading LLMs for their Ability to Replicate a Human Survey with Synthetic Data","zh_title":"随机鹦鹉还是和谐合唱？测试五大领先LLM用合成数据复现人类调查的能力","primary_category":"cs.CY","date":"2026-08-07","score":10,"bucket":"selected","tags":["LLM仿真","调查复现","可靠性评估"],"rubric_hits":["A1","A2","A4","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2603.00059","has_summary":true},{"id":"2608.06085","title":"Signal or Spurious Cue? A Randomized Audit of Survey-Country Metadata in LLM Social Inference","zh_title":"信号还是虚假线索？一项关于LLM社会推断中调查国家元数据的随机审计","primary_category":"cs.AI","date":"2026-08-07","score":9,"bucket":"selected","tags":["LLM仿真","调查回答预测","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.06085","has_summary":true},{"id":"2608.06115","title":"Mind the Gaps: Mixture-of-Minds for Human Simulation","zh_title":"注意差距：用于人类仿真的思维混合模型","primary_category":"cs.AI","date":"2026-08-07","score":9,"bucket":"selected","tags":["人类仿真","调查预测","个体异质性"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.06115","has_summary":true},{"id":"2608.06151","title":"Reducing belief in conspiracy theories as they unfold using large language models","zh_title":"使用大语言模型减少实时阴谋论信念","primary_category":"cs.HC","date":"2026-08-07","score":9,"bucket":"selected","tags":["LLM人类仿真","阴谋论干预","行为实验"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.06151","has_summary":true},{"id":"2608.05178","title":"Who Gets Access? Global Region and Academic Status Bias in AI-Generated Academic Gatekeeping Scenarios","zh_title":"谁获得访问权？AI生成学术把关场景中的全球区域与学术地位偏见","primary_category":"cs.CY","date":"2026-08-07","score":7,"bucket":"pending","tags":["LLM仿真","学术把关","偏见审计"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2608.05178","has_summary":true},{"id":"2608.05583","title":"The Judgment-Consequence Gap: LLM Moral Reasoning in Healthcare Decisions","zh_title":"判断-后果差距：医疗决策中大语言模型的道德推理","primary_category":"cs.CY","date":"2026-08-07","score":7,"bucket":"pending","tags":["LLM仿真","道德决策","人类对照"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.05583","has_summary":true},{"id":"2608.00023","title":"Role Steering of Language Models for Social Simulations","zh_title":"面向社会模拟的语言模型角色引导","primary_category":"cs.CL","date":"2026-08-07","score":5,"bucket":"other","tags":["社会模拟","角色引导","LLM代理"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.00023","has_summary":false},{"id":"2608.05166","title":"Conditional Cognitive Biases in LLMs: How Biased User Turns Modulate In-Context Reasoning","zh_title":"大语言模型中的条件认知偏差：有偏用户轮次如何调节上下文推理","primary_category":"cs.CL","date":"2026-08-07","score":5,"bucket":"other","tags":["认知偏差","LLM评估","上下文影响"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.05166","has_summary":false},{"id":"2608.05576","title":"Where Models Converge and Humans Diverge: A Coverage Framework for Distributional Pluralism in Open-Ended Generation","zh_title":"模型趋同与人类发散之处：开放式生成中分布多元性的覆盖框架","primary_category":"cs.CL","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM生成分布","文化覆盖度","人类写作对比"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.05576","has_summary":false},{"id":"2608.05630","title":"Human-Like Anaphor Resolution in Large Language Models","zh_title":"大型语言模型中类人指代消解研究","primary_category":"cs.CL","date":"2026-08-07","score":5,"bucket":"other","tags":["指代消解","认知对齐","语言模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.05630","has_summary":false},{"id":"2608.05519","title":"EcoAgent-Bench: Evaluating Economic Decision-Making in Budget-Constrained LLM Agents","zh_title":"EcoAgent-Bench：评估预算受限LLM智能体的经济决策","primary_category":"cs.AI","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM智能体","经济决策","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.05519","has_summary":false},{"id":"2608.05367","title":"Counterfactual Analysis via Large Language Models","zh_title":"基于大语言模型的反事实分析","primary_category":"cs.AI","date":"2026-08-07","score":5,"bucket":"other","tags":["反事实分析","LLM仿真","在线借贷"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.05367","has_summary":false},{"id":"2608.05864","title":"Seeing Is Not Deciding: Can Multimodal LLMs Act as Effective CEOs?","zh_title":"眼见不为实：多模态大语言模型能胜任CEO吗？","primary_category":"cs.AI","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM决策模拟","多模态智能体","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.05864","has_summary":false},{"id":"2608.06020","title":"From Economic Agents to Agentic Economies: A Systems Blueprint for Economic World Models","zh_title":"从经济主体到主体经济：经济世界模型的系统蓝图","primary_category":"cs.AI","date":"2026-08-07","score":5,"bucket":"other","tags":["经济世界模型","LLM智能体","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06020","has_summary":false},{"id":"2608.06108","title":"Evaluating Investment Logic in Large Language Models: A Real-World Benchmark Towards Personalzied Financial Agents","zh_title":"评估大语言模型中的投资逻辑：面向个性化金融智能体的真实世界基准","primary_category":"cs.AI","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM评估","金融智能体","决策模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.06108","has_summary":false},{"id":"2608.05172","title":"Estimating time spent on work tasks","zh_title":"估算工作任务耗时","primary_category":"cs.CY","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM标注","任务时间估计","AI暴露度"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.05172","has_summary":false},{"id":"2608.05180","title":"The Nuclear Decision-Making Benchmark: Evaluating Frontier LLMs on Nuclear Tendencies","zh_title":"核决策基准：评估前沿大语言模型的核倾向","primary_category":"cs.CY","date":"2026-08-07","score":5,"bucket":"other","tags":["LLM评估","核决策","模型行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.05180","has_summary":false},{"id":"2608.03067","title":"Activation-Guided Neuron Intervention to Induce Alzheimer's-Related Computational Language Phenotypes in a Large Language Model","zh_title":"激活引导神经元干预以诱导大语言模型产生阿尔茨海默症相关计算语言表型","primary_category":"cs.CL","date":"2026-08-07","score":0,"bucket":"other","tags":["阿尔茨海默症","语言检测","模型编辑"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03067","has_summary":false},{"id":"2608.04549","title":"EuroExec: Frontier Language Models Fall Short of Expert Judgment on European Executive Decision Tasks","zh_title":"EuroExec：前沿语言模型在欧洲行政决策任务上不及专家判断","primary_category":"cs.CL","date":"2026-08-07","score":0,"bucket":"other","tags":["LLM评测","专家基准","行政决策"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04549","has_summary":false},{"id":"2608.05993","title":"Clinical Communication Processing with Models Trained on LLM-Generated Synthetic Data: A Structured Survey and Novel Application Case Studies","zh_title":"基于LLM生成合成数据训练的临床沟通处理：结构化综述与新应用案例","primary_category":"cs.CL","date":"2026-08-07","score":0,"bucket":"other","tags":["临床NLP","合成数据","数据增强"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.05993","has_summary":false},{"id":"2608.06027","title":"FormBharo: Designing and Evaluating a Voice Agent for Conversational Form Filling in Rural India","zh_title":"FormBharo：为印度农村设计的对话式表单填写语音代理","primary_category":"cs.CL","date":"2026-08-07","score":0,"bucket":"other","tags":["语音代理","表单填写","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.06027","has_summary":false},{"id":"2608.05889","title":"The em-dash em-beds in Congress: A population-level rise in em-dash frequency in U.S. congressional press releases at the dawn of the large-language-model era, 2021-2025","zh_title":"国会中的破折号嵌入：大语言模型时代初期美国国会新闻稿中破折号频率的群体性上升（2021-2025）","primary_category":"cs.DL","date":"2026-08-07","score":0,"bucket":"other","tags":["LLM写作痕迹","文本风格分析","国会新闻稿"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.05889","has_summary":false},{"id":"2608.05710","title":"Shaping Human-AI Interactions to Provide Improvement Pathways and Balance Competing Objectives","zh_title":"塑造人机交互以提供改进路径并平衡竞争目标","primary_category":"cs.AI","date":"2026-08-07","score":0,"bucket":"other","tags":["人机交互","多智能体系统","机器学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05710","has_summary":false},{"id":"2608.05778","title":"When Do Prompt-Side Agent Playbooks Transfer? Accuracy, Cost, and Runtime Shift in Agent Deployment","zh_title":"提示侧智能体操作手册何时可迁移？智能体部署中的准确性、成本与运行时偏移","primary_category":"cs.AI","date":"2026-08-07","score":0,"bucket":"other","tags":["多智能体系统","工具调用","迁移学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05778","has_summary":false},{"id":"2608.05171","title":"Beyond Information Retrieval: Generative AI as an Epistemic Arbiter to Enhance Collaborative Problem-Solving","zh_title":"超越信息检索：生成式AI作为认知仲裁者以增强协作问题解决","primary_category":"cs.CY","date":"2026-08-07","score":0,"bucket":"other","tags":["协作问题解决","生成式AI","教育技术"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05171","has_summary":false},{"id":"2608.06202","title":"What Current AI Benchmarks Leave Unmeasured: Modality, Search, Citations, and Implications (for Safety Evaluations)","zh_title":"当前AI基准测试未测量的内容：模态、搜索、引用及其对安全评估的影响","primary_category":"cs.HC","date":"2026-08-07","score":0,"bucket":"other","tags":["LLM评测","基准测试","安全评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06202","has_summary":false},{"id":"2608.06353","title":"Resourced Authority A Mechanism-Design Model for Participatory Governance of Deployed AI Agents","zh_title":"资源化权威：面向已部署AI代理参与式治理的机制设计模型","primary_category":"cs.GT","date":"2026-08-07","score":0,"bucket":"other","tags":["机制设计","多智能体系统","AI治理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06353","has_summary":false},{"id":"2608.06166","title":"What out-of-the-box LLMs can(t) do in law? A Turing test in Italian exams for lawyers, judges and notaries","zh_title":"开箱即用的大语言模型在法律领域能（不能）做什么？意大利律师、法官和公证人考试的图灵测试","primary_category":"cs.CY","date":"2026-08-07","score":0,"bucket":"other","tags":["LLM评测","法律考试","图灵测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.06166","has_summary":false},{"id":"2608.06322","title":"From Precision Medicine to Precision Education: A Vision for AI-Powered Student Digital Twins, Preventive Student Success, and Career-Aligned Academic Pathways","zh_title":"从精准医疗到精准教育：AI驱动的学生数字孪生、预防性学生成功与职业对齐学术路径的愿景","primary_category":"cs.CY","date":"2026-08-07","score":0,"bucket":"other","tags":["精准教育","学生数字孪生","学习分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.06322","has_summary":false},{"id":"2608.04205","title":"MatrAIx: Simulating the World with 8.3 Billion Persona Agents","zh_title":"MatrAIx：用83亿人格代理模拟世界","primary_category":"cs.AI","date":"2026-08-06","score":9,"bucket":"selected","tags":["LLM人类仿真","大规模人格代理","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.04205","has_summary":true},{"id":"2608.04020","title":"Artificial Institutions: How Institutional Design Shapes LLM Simulations","zh_title":"人工制度：制度设计如何塑造LLM仿真","primary_category":"cs.CY","date":"2026-08-06","score":8,"bucket":"selected","tags":["LLM仿真","市场实验","制度设计"],"rubric_hits":["A3","B2"],"abs_url":"https://arxiv.org/abs/2608.04020","has_summary":true},{"id":"2608.02491","title":"Long-term Measurements: Towards a Longitudinal Understanding of Human-AI Interactions","zh_title":"长期测量：迈向人机交互的纵向理解","primary_category":"cs.AI","date":"2026-08-06","score":5,"bucket":"other","tags":["人机交互","纵向研究","行为变化"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.02491","has_summary":false},{"id":"2608.04095","title":"FinPerMA: A Theory-Informed, Event-Grounded Personalized-Memory Benchmark for LLM Agents","zh_title":"FinPerMA：一个理论驱动、事件锚定的个性化记忆基准，用于评估LLM智能体","primary_category":"cs.AI","date":"2026-08-06","score":5,"bucket":"other","tags":["LLM智能体","个性化记忆","金融行为模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.04095","has_summary":false},{"id":"2608.04056","title":"Learning Sexism Detection Using Multi-Agent Perspectivist Preference Optimization","zh_title":"使用多智能体视角偏好优化学习性别歧视检测","primary_category":"cs.CL","date":"2026-08-06","score":5,"bucket":"other","tags":["多智能体","标注分歧","偏好优化"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.04056","has_summary":false},{"id":"2608.04507","title":"Emergence of Reputation-Based Cooperation in LLM Agents","zh_title":"基于声誉的合作在LLM智能体中的涌现","primary_category":"cs.MA","date":"2026-08-06","score":5,"bucket":"other","tags":["LLM智能体","间接互惠","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.04507","has_summary":false},{"id":"2607.21597","title":"Risk Is Not the Target: A Monotonic Framework for Evaluating Wildfire Operational Risk Signals","zh_title":"风险不是目标：评估野火运营风险信号的单调框架","primary_category":"cs.AI","date":"2026-08-06","score":2,"bucket":"other","tags":["野火风险","多智能体系统","评估框架"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21597","has_summary":false},{"id":"2608.02046","title":"CompanionBench: A Theory-Anchored, Real-World-Grounded Benchmark for AI Emotional Companionship","zh_title":"CompanionBench：一个理论锚定、真实世界基准的AI情感陪伴评测基准","primary_category":"cs.CL","date":"2026-08-06","score":0,"bucket":"other","tags":["情感陪伴","基准评测","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.02046","has_summary":false},{"id":"2608.03722","title":"When Outputs Disperse, Does Epistemic Revision Follow? A Black-Box Diagnostic for Machine Collectives","zh_title":"当输出分散时，认知修正会随之而来吗？一种面向机器集体的黑盒诊断方法","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体系统","认知诊断","LLM集体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03722","has_summary":false},{"id":"2608.04663","title":"Calibrating Artificial Guilt: Neurally Grounded Reward Shaping for Prosocial Multi-Agent Reinforcement Learning","zh_title":"校准人工内疚：基于神经基础的多智能体亲社会奖励塑形","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体强化学习","奖励塑形","神经数据"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04663","has_summary":false},{"id":"2608.04697","title":"Traceable LLM-Generated Hazard Scenarios for Operational Safety Analysis of Aviation Systems Using ASRS Reports","zh_title":"基于ASRS报告的可溯源LLM生成危险场景用于航空系统运行安全分析","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["航空安全","场景生成","LLM评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04697","has_summary":false},{"id":"2608.04735","title":"Chain-of-Thought Monitoring Can Be Unreliable in Implicit-Influence Settings","zh_title":"思维链监控在隐性影响设置下可能不可靠","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["思维链监控","AI安全","模型评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04735","has_summary":false},{"id":"2608.05030","title":"From Score Matrices to Football-Aware Match-State Simulation: An Auditable LLM Harness for Exact-Score Reranking","zh_title":"从得分矩阵到足球感知的比赛状态模拟：一种可审计的LLM harness用于精确比分重排序","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["足球预测","混合模型","LLM应用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05030","has_summary":false},{"id":"2608.05086","title":"Item Response Theory for AI Safety","zh_title":"项目反应理论用于AI安全","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["AI安全","基准评测","心理测量"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.05086","has_summary":false},{"id":"2608.05107","title":"CoPlan: A Trustworthy Co-Intelligence Interface for Care Planning through Role-Based Contestable Argument Graphs","zh_title":"CoPlan：基于角色可争议论证图的可信协同智能护理规划接口","primary_category":"cs.AI","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体系统","人机协同","护理规划"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05107","has_summary":false},{"id":"2608.04148","title":"AgentForge: An Immersive Role-Playing Platform for Learning Agentic Software Engineering","zh_title":"AgentForge：一个用于学习智能体软件工程的沉浸式角色扮演平台","primary_category":"cs.SE","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体系统","软件工程教育","角色扮演"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04148","has_summary":false},{"id":"2608.04591","title":"When Absence Is Evidence: Evaluating Completeness-Sensitive Negative Reasoning in Large Language Models","zh_title":"当缺失成为证据：评估大语言模型中的完整性敏感否定推理","primary_category":"cs.CL","date":"2026-08-06","score":0,"bucket":"other","tags":["否定推理","NLP评测","证据完整性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04591","has_summary":false},{"id":"2608.04714","title":"What We Observe as LLM Behavior Can Be a Side-effect of Inference Backend","zh_title":"我们观察到的LLM行为可能是推理后端的副作用","primary_category":"cs.SE","date":"2026-08-06","score":0,"bucket":"other","tags":["LLM评测","推理后端","基准分数"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04714","has_summary":false},{"id":"2608.04980","title":"Protoreasoning in Tiny Transformers","zh_title":"微型Transformer中的原型推理","primary_category":"cs.CL","date":"2026-08-06","score":0,"bucket":"other","tags":["推理机制","Dyck语言","小型模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04980","has_summary":false},{"id":"2608.05015","title":"Revealed Rationality: Label-Free Evaluation and Regularization from Representation Theorems","zh_title":"揭示理性：基于表示定理的无标签评估与正则化","primary_category":"econ.TH","date":"2026-08-06","score":0,"bucket":"other","tags":["决策论","LLM评估","理性公理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.05015","has_summary":false},{"id":"2608.05026","title":"ArtAnno: Annotating Implicit Semantics in Artworks through LLM Agent-Driven Bidirectional Human-AI Augmentation","zh_title":"ArtAnno：通过LLM智能体驱动的双向人机增强标注艺术品中的隐含语义","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体系统","艺术品标注","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05026","has_summary":false},{"id":"2608.05064","title":"Provable Limits and Certified Deferral for Verbalized Uncertainty in Small Language Models","zh_title":"小语言模型口头不确定性的可证明极限与认证推迟","primary_category":"cs.CL","date":"2026-08-06","score":0,"bucket":"other","tags":["模型校准","风险控制","选择性预测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.05064","has_summary":false},{"id":"2608.04120","title":"Echoes in the Sky: Computational Thematic Analysis of Online Public Discourse on Bluesky Across Trump's Reelection","zh_title":"天空中的回声：特朗普连任期间Bluesky在线公共话语的计算主题分析","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["社交媒体分析","主题建模","LLM辅助聚类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.04120","has_summary":false},{"id":"2608.04166","title":"Enacting Constructive Conflicts with AI Agents to Enhance Reconsideration among Novice Interaction Designers","zh_title":"利用AI代理实施建设性冲突以增强新手交互设计师的反思","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["人机交互","设计教育","AI代理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.04166","has_summary":false},{"id":"2608.04416","title":"Preference-Driven Online Adaptation for Personalized Interaction Initiation in Proactive AI Assistants","zh_title":"主动式AI助手中基于偏好驱动的在线自适应个性化交互启动","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["主动AI助手","个性化交互","在线适应"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.04416","has_summary":false},{"id":"2608.04831","title":"Investigating Click Behaviors On Google Search Result Pages That Produce an AI Overview","zh_title":"调查谷歌搜索结果页中产生AI概述的点击行为","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["用户行为分析","搜索引擎","AI概述"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04831","has_summary":false},{"id":"2608.04951","title":"Reply, Delete, or Ignore? Examining How Content Creators Perceive and Select Comment Moderation Strategies","zh_title":"回复、删除还是忽略？考察内容创作者如何感知和选择评论管理策略","primary_category":"cs.HC","date":"2026-08-06","score":0,"bucket":"other","tags":["内容创作者","评论管理","在线安全"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.04951","has_summary":false},{"id":"2608.04774","title":"Decentralization of Agenda-Setting Power and Domain-Selective Bridging: Algorithm Design Beyond the Echo Chamber Debate","zh_title":"议程设置权力去中心化与领域选择性桥接：超越回音室辩论的算法设计","primary_category":"cs.CY","date":"2026-08-06","score":0,"bucket":"other","tags":["多智能体仿真","算法设计","回音室"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04774","has_summary":false},{"id":"2608.05008","title":"The Beginning of ChatGPT Ads","zh_title":"ChatGPT广告的开端","primary_category":"cs.CY","date":"2026-08-06","score":0,"bucket":"other","tags":["广告审计","多智能体","算法公平"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.05008","has_summary":false},{"id":"2608.04318","title":"Responsibility in Multi-Agent Sequential Decision-Making: Comparing Human Judgments to Formal Models of Causal Attribution","zh_title":"多智能体序贯决策中的责任归属：人类判断与因果归因形式模型的比较","primary_category":"cs.MA","date":"2026-08-06","score":0,"bucket":"other","tags":["责任归因","多智能体决策","人类判断"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04318","has_summary":false},{"id":"2608.04524","title":"ODRA: Synthesizing Cognitive Behavioral Therapy Sessions with Structured Chain-Of-Thought and Dynamic Patient Resistance","zh_title":"ODRA：利用结构化思维链与动态患者阻抗合成认知行为治疗会话","primary_category":"cs.CL","date":"2026-08-06","score":0,"bucket":"other","tags":["对话生成","认知行为治疗","多智能体"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.04524","has_summary":false},{"id":"2608.04198","title":"Does generative AI narrow education-based productivity gaps? Evidence from a randomized experiment","zh_title":"生成式AI是否缩小了基于教育的生产力差距？来自随机实验的证据","primary_category":"econ.GN","date":"2026-08-06","score":0,"bucket":"other","tags":["生产力差距","随机实验","AI辅助"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04198","has_summary":false},{"id":"2608.02758","title":"Everyone Conforms, No One Believes: Pluralistic Ignorance in LLM Agent Populations","zh_title":"人人从众，无人相信：LLM智能体群体中的多元无知","primary_category":"cs.MA","date":"2026-08-05","score":9,"bucket":"selected","tags":["LLM仿真","多元无知","社会规范"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.02758","has_summary":true},{"id":"2602.04000","title":"After Talking with 1,000 Personas: Learning Preference-Aligned Proactive Assistants From Large-Scale Persona Interactions","zh_title":"与1000个角色对话后：从大规模角色交互中学习偏好对齐的主动助手","primary_category":"cs.HC","date":"2026-08-05","score":7,"bucket":"pending","tags":["LLM仿真","人类行为模拟","偏好学习"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2602.04000","has_summary":true},{"id":"2608.03239","title":"Relational Priors as Convergence Pressure in LLM-Based Multi-Agent Systems","zh_title":"基于LLM的多智能体系统中关系先验作为收敛压力","primary_category":"cs.CL","date":"2026-08-05","score":5,"bucket":"other","tags":["多智能体系统","社会模拟","关系先验"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.03239","has_summary":false},{"id":"2608.03532","title":"Cross-Lingual Bias in Large Language Models: A Comparative Analysis of English and Swahili","zh_title":"大语言模型的跨语言偏见：英语与斯瓦希里语的比较分析","primary_category":"cs.CL","date":"2026-08-05","score":5,"bucket":"other","tags":["偏见评估","跨语言","安全对齐"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.03532","has_summary":false},{"id":"2608.03659","title":"How Closely Do LLM Reviews Align with Human Peer Review?","zh_title":"LLM评审与人类同行评审的一致性有多高？","primary_category":"cs.CL","date":"2026-08-05","score":5,"bucket":"other","tags":["LLM评审","同行评审","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.03659","has_summary":false},{"id":"2608.03206","title":"EduClaw-Bench: A Long-Horizon Benchmark for Pedagogical LLM Agents with Simulated Learners","zh_title":"EduClaw-Bench：面向教学型LLM代理与模拟学习者的长周期基准测试","primary_category":"cs.CY","date":"2026-08-05","score":5,"bucket":"other","tags":["LLM教学代理","模拟学习者","知识追踪"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.03206","has_summary":false},{"id":"2608.03585","title":"From Social Coding to Agentic Coding: Productivity and Relational Reconfiguration in Open-Source Communities","zh_title":"从社交编码到智能体编码：开源社区中的生产力与关系重构","primary_category":"cs.AI","date":"2026-08-05","score":5,"bucket":"other","tags":["LLM多智能体仿真","开源社区","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.03585","has_summary":false},{"id":"2608.02827","title":"Emergence of Biased Consensus in Multi-Agent LLM Debates","zh_title":"多智能体LLM辩论中偏见共识的涌现","primary_category":"cs.MA","date":"2026-08-05","score":5,"bucket":"other","tags":["多智能体辩论","社会模拟","偏见涌现"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.02827","has_summary":false},{"id":"2608.03416","title":"AI World Cup 2026: Benchmarking Large Language Models for End-to-End Football Tournament Prediction","zh_title":"AI世界杯2026：评估大语言模型端到端足球赛事预测能力","primary_category":"cs.AI","date":"2026-08-05","score":3,"bucket":"other","tags":["LLM预测","足球赛事","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03416","has_summary":false},{"id":"2608.01679","title":"When Memory Becomes Authority: Benchmarking Authority Collapse at the Memory Consolidation Boundary","zh_title":"当记忆成为权威：在记忆巩固边界上基准测试权威坍塌","primary_category":"cs.AI","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM智能体","记忆巩固","权威坍塌"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01679","has_summary":false},{"id":"2608.00151","title":"Optimising for Flourishing: Flourishing Metrics and Return on Flourishing as Success Criteria for Artificial Intelligence and Post-AGI Economic Systems","zh_title":"优化繁荣：繁荣指标与繁荣回报作为人工智能及后AGI经济系统的成功标准","primary_category":"cs.CY","date":"2026-08-05","score":0,"bucket":"other","tags":["AI伦理","经济系统","人类福祉"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00151","has_summary":false},{"id":"2608.01366","title":"Asking Questions the Right Way: A Multi-Agent Conversational System for Prompt Formulation in Complex Task Resolution","zh_title":"以正确方式提问：面向复杂任务解决的提示词构建多智能体对话系统","primary_category":"cs.MA","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体系统","提示词优化","人机交互"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01366","has_summary":false},{"id":"2608.02807","title":"Learning a Vector-Symbolic Model for Socio-Cultural Tasks","zh_title":"学习用于社会文化任务的向量符号模型","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["认知建模","ACT-R架构","内隐联想测验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02807","has_summary":false},{"id":"2608.02941","title":"Aligned in Form, Not in Meaning: The Comprehension - Containment Decoupling of LLM Safety in Low-Resource Bangla Derogatory Speech","zh_title":"形式对齐，意义未对齐：低资源孟加拉语贬损言论中LLM安全的理解-抑制解耦","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM安全","低资源语言","理解-抑制解耦"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02941","has_summary":false},{"id":"2608.02966","title":"Every Wrong Answer Counts: Option-Level Psychometrics for LLM Multiple-Choice Benchmarks","zh_title":"每个错误答案都算数：LLM多选题基准的选项级心理测量学","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM评测","心理测量模型","多选题分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02966","has_summary":false},{"id":"2608.03035","title":"Language Models Encode the Contextual Truth of Propositions","zh_title":"语言模型编码命题的上下文真值","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM表征","多智能体","真值编码"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03035","has_summary":false},{"id":"2608.03038","title":"Beyond Accuracy: A Multidimensional Evaluation of Statistical Reasoning in Large Language Models","zh_title":"超越准确率：大语言模型统计推理的多维评估","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM评测","统计推理","多维分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03038","has_summary":false},{"id":"2608.03233","title":"On the Diversity of Analogy Making in Large Language Models","zh_title":"大语言模型类比生成多样性的研究","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["类比生成","输出多样性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03233","has_summary":false},{"id":"2608.03340","title":"Benchmarking the Benchmarks: Testing the Predictive Validity of Commonsense Benchmarks","zh_title":"基准测试的基准：检验常识基准的预测效度","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["基准评测","常识推理","预测效度"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03340","has_summary":false},{"id":"2608.03358","title":"ArtECulture: Benchmarking Culture-Conditioned Visual Emotion Understanding in Multimodal Large Language Models","zh_title":"ArtECulture：多模态大语言模型中文化条件化视觉情感理解的基准测试","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["多模态大模型","情感理解","文化差异"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03358","has_summary":false},{"id":"2608.03388","title":"Don't Let Me Ask for It: LLMs Show Deficiencies in Active Multi-Turn Information Acquisition for Abductive Inference","zh_title":"别让我问：大语言模型在溯因推理的主动多轮信息获取中表现不足","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM推理","多轮交互","溯因推理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03388","has_summary":false},{"id":"2608.03810","title":"VIBE: A VAD-Informed Benchmark for Entity-Centered Affective Profiling of Large Language Model Outputs","zh_title":"VIBE：基于VAD的实体中心情感画像基准，用于大语言模型输出","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["情感分析","基准测试","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03810","has_summary":false},{"id":"2608.04003","title":"PAST-Bench: Benchmarking the Foundations of Recursive Self-Improvement in Personal Agents","zh_title":"PAST-Bench：个人智能体中递归自我改进基础的基准测试","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体系统","基准测试","递归自我改进"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04003","has_summary":false},{"id":"2608.04008","title":"WorldCup Arena: Prospective, Leakage-Free Evaluation of Frontier LLMs on a Live Tournament","zh_title":"WorldCup Arena：前沿大语言模型在实时赛事中的前瞻性无泄漏评估","primary_category":"cs.CL","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM预测","体育赛事","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.04008","has_summary":false},{"id":"2608.02618","title":"Beyond the Hivemind: Escaping LLM Homogeneity via Meta-Persona Anchoring and Sequential Temperature Scaling","zh_title":"超越蜂群思维：通过元人格锚定与顺序温度缩放逃离LLM同质化","primary_category":"cs.AI","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM多样性","文本生成","语义坍缩"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02618","has_summary":false},{"id":"2608.03700","title":"When Agents Learn to Be You: Benchmarking Privacy Leakage, Impersonation Risk, and Defenses in Persona Skills","zh_title":"当智能体学会成为你：角色技能中的隐私泄露、冒充风险与防御基准测试","primary_category":"cs.CR","date":"2026-08-05","score":0,"bucket":"other","tags":["角色扮演","隐私安全","智能体评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.03700","has_summary":false},{"id":"2608.03874","title":"ContinualSkillBench: Can LLM Agents Truly Evolve Their Capabilities?","zh_title":"ContinualSkillBench：LLM智能体能否真正进化其能力？","primary_category":"cs.AI","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体","技能学习","评测基准"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03874","has_summary":false},{"id":"2608.03433","title":"Cross-cultural evaluation of taste-sound correspondences in AI-generated music","zh_title":"AI生成音乐中味觉-声音对应关系的跨文化评估","primary_category":"cs.HC","date":"2026-08-05","score":0,"bucket":"other","tags":["音乐生成","跨文化感知","味觉-声音对应"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.03433","has_summary":false},{"id":"2608.03462","title":"When AI Joins the Team! A Model of How AI Adoption Relates To Social Patterns in Software Engineering Teams","zh_title":"当AI加入团队：AI采用如何与软件工程团队中的社会模式相关联的模型","primary_category":"cs.SE","date":"2026-08-05","score":0,"bucket":"other","tags":["AI辅助开发","团队协作","社区异味"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03462","has_summary":false},{"id":"2608.03500","title":"LLM-Assisted Review Prioritization for German Statutory Health Insurance Websites: A Multi-Stage Corpus Audit","zh_title":"基于LLM的德国法定健康保险网站审核优先级辅助：多阶段语料审计","primary_category":"cs.CY","date":"2026-08-05","score":0,"bucket":"other","tags":["LLM辅助审核","网站内容审计","多阶段工作流"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03500","has_summary":false},{"id":"2608.03800","title":"Autoreflection: How Agentic Strange Loops Turn Human Culture into AI Infrastructure","zh_title":"自反身性：代理式奇异循环如何将人类文化转化为AI基础设施","primary_category":"cs.CY","date":"2026-08-05","score":0,"bucket":"other","tags":["AI代理","自反身性","社交平台"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.03800","has_summary":false},{"id":"2608.03904","title":"Why do we need social singularity? A mechanism-based critique of gradual scenarios in AI existential-risk discourse","zh_title":"为什么我们需要社会奇点？对AI存在风险话语中渐进情景的机制性批判","primary_category":"cs.CY","date":"2026-08-05","score":0,"bucket":"other","tags":["AI风险","社会机制","技术奇点"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03904","has_summary":false},{"id":"2608.03973","title":"When AI Wears Many Hats: The Role of Generative Artificial Intelligence in Marketing Education","zh_title":"当AI身兼多职：生成式人工智能在营销教育中的作用","primary_category":"cs.CY","date":"2026-08-05","score":0,"bucket":"other","tags":["营销教育","生成式AI","角色理论"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.03973","has_summary":false},{"id":"2608.03114","title":"Optimal Liability Design for Medical AI","zh_title":"医疗人工智能的最优责任设计","primary_category":"econ.TH","date":"2026-08-05","score":0,"bucket":"other","tags":["医疗AI","责任设计","委托代理模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03114","has_summary":false},{"id":"2608.03648","title":"Group Perspective Matters: Regulating Debate Relationships Can Mitigate Blind Conformity in Multi-Agent Debate","zh_title":"群体视角至关重要：调节辩论关系可缓解多智能体辩论中的盲从","primary_category":"cs.MA","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体辩论","盲从缓解","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03648","has_summary":false},{"id":"2608.02677","title":"When Policies Change Probabilities: Modular Decision-Making for LLM Code Review","zh_title":"当策略改变概率：LLM代码审查的模块化决策","primary_category":"cs.SE","date":"2026-08-05","score":0,"bucket":"other","tags":["代码审查","多智能体","决策分离"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02677","has_summary":false},{"id":"2608.03272","title":"Attacking and Defending Multi-Agent Collaborative Filtering Systems Through Connectivity","zh_title":"通过连接性攻击与防御多智能体协同过滤系统","primary_category":"cs.IR","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体系统","协同过滤","对抗攻击"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03272","has_summary":false},{"id":"2608.03910","title":"Socially Grounded Agentic AI: Coordinating Plural Perspectives through Social Theory","zh_title":"基于社会理论的智能体AI：通过社会理论协调多元视角","primary_category":"cs.AI","date":"2026-08-05","score":0,"bucket":"other","tags":["多智能体系统","价值对齐","社会理论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03910","has_summary":false},{"id":"2608.02877","title":"GoT-CD: Graph-of-Thoughts Causal Discovery and the Fragility of Post-hoc Path-Specific Fairness Audits","zh_title":"GoT-CD：思维图因果发现与事后路径特定公平性审计的脆弱性","primary_category":"cs.LG","date":"2026-08-05","score":0,"bucket":"other","tags":["因果发现","公平性审计","图推理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02877","has_summary":false},{"id":"2608.03085","title":"Causal Inference with Unstructured Outcomes","zh_title":"非结构化结局的因果推断","primary_category":"stat.ML","date":"2026-08-05","score":0,"bucket":"other","tags":["因果推断","非结构化数据","机器学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.03085","has_summary":false},{"id":"2608.03382","title":"LLM-Derived Priors for Thompson Sampling in Cold-Start Comment Recommendation","zh_title":"基于大语言模型先验的冷启动评论推荐汤普森采样","primary_category":"cs.IR","date":"2026-08-05","score":0,"bucket":"other","tags":["推荐系统","多臂老虎机","冷启动"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03382","has_summary":false},{"id":"2608.03606","title":"Learning Clinical-Trial Strategy: Offline Policy Training for Decision Agents","zh_title":"学习临床试验策略：决策智能体的离线策略训练","primary_category":"cs.AI","date":"2026-08-05","score":0,"bucket":"other","tags":["离线强化学习","临床试验决策","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03606","has_summary":false},{"id":"2608.03153","title":"Does the Gender Wage Gap Originate at Labor Market Entry? Evidence from South Korea","zh_title":"性别工资差距是否源于劳动力市场进入？来自韩国的证据","primary_category":"econ.GN","date":"2026-08-05","score":0,"bucket":"other","tags":["性别工资差距","劳动经济学","韩国"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2608.03153","has_summary":false},{"id":"2608.02909","title":"When Predictions Become Regressors: A Split-Sample Correction for Biases in Downstream Inference","zh_title":"当预测成为回归变量：下游推断偏差的分割样本校正","primary_category":"econ.EM","date":"2026-08-05","score":0,"bucket":"other","tags":["测量误差","工具变量","政治学方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02909","has_summary":false},{"id":"2608.03125","title":"Performance appraisal promotes cooperation in spatial public goods games","zh_title":"绩效评估促进空间公共物品博弈中的合作","primary_category":"physics.soc-ph","date":"2026-08-05","score":0,"bucket":"other","tags":["演化博弈","合作涌现","空间公共物品博弈"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.03125","has_summary":false},{"id":"2607.29334","title":"The persuasive power of large language models does not depend on their perceived national origin","zh_title":"大语言模型的说服力不依赖于其感知的国家来源","primary_category":"cs.HC","date":"2026-08-04","score":9,"bucket":"selected","tags":["LLM仿真","人类被试替代","说服实验"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.29334","has_summary":true},{"id":"2608.01212","title":"Do Humans Bargain Differently with AI? Evidence from Alternating-Offer Games","zh_title":"人类与AI的讨价还价行为不同吗？来自交替报价博弈的证据","primary_category":"econ.GN","date":"2026-08-04","score":9,"bucket":"selected","tags":["LLM仿真","行为博弈","人机交互"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01212","has_summary":true},{"id":"2608.01607","title":"AI Financial Advice: Supply, Demand, and Life Cycle Implications","zh_title":"人工智能财务建议：供给、需求与生命周期影响","primary_category":"econ.GN","date":"2026-08-04","score":9,"bucket":"selected","tags":["LLM仿真","财务决策","人类行为对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.01607","has_summary":true},{"id":"2608.01204","title":"ShiJianBench: From Dialogue to Decision for Long-Horizon Evaluation of Investment Advisors","zh_title":"ShiJianBench：从对话到决策的长期投资顾问评估","primary_category":"cs.CL","date":"2026-08-04","score":8,"bucket":"selected","tags":["LLM仿真","投资者行为","人类数据校准"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01204","has_summary":true},{"id":"2608.01458","title":"PALMs: Using Multi Construct-Grounded Rationales for Modeling Population Preferences in LLMs","zh_title":"PALMs：使用多构念基础理由建模大语言模型中的人口偏好","primary_category":"cs.CL","date":"2026-08-04","score":8,"bucket":"selected","tags":["人口仿真","文化对齐","偏好建模"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01458","has_summary":true},{"id":"2608.01629","title":"Human-LLM Alignment in Language Attitudes Toward Non-Native Japanese","zh_title":"人类与LLM对非母语日语语言态度的一致性","primary_category":"cs.CL","date":"2026-08-04","score":8,"bucket":"selected","tags":["语言态度","人类仿真","偏差审计"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2608.01629","has_summary":true},{"id":"2608.00979","title":"Passing Coarse Marginal Checks Can Be Cheap: Persona Mixtures and Imprecise Treatment-Response Estimates in an LLM Persona Panel","zh_title":"通过粗粒度边际检查可能很廉价：LLM角色面板中的角色混合与不精确的处理效应估计","primary_category":"cs.AI","date":"2026-08-04","score":8,"bucket":"selected","tags":["LLM仿真","行为博弈","算法保真度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2608.00979","has_summary":true},{"id":"2608.01193","title":"Humans Are More Diverse: Frontier LLMs Show Extreme Policies in Idealised AI Development Races","zh_title":"人类更多样：前沿大语言模型在理想化AI发展竞赛中表现出极端策略","primary_category":"cs.AI","date":"2026-08-04","score":8,"bucket":"selected","tags":["LLM行为仿真","博弈实验","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01193","has_summary":true},{"id":"2607.27553","title":"AI and Its Impact on Creativity and Diversity: An Empirical Study of LLM-Generated Product Ideas","zh_title":"AI对创造力与多样性的影响：LLM生成产品创意的实证研究","primary_category":"cs.AI","date":"2026-08-04","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","产品创新"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.27553","has_summary":true},{"id":"2608.01181","title":"Talking to Digital Twins: Selective Disclosure and Belief Measurement in Financial Social Media","zh_title":"与数字孪生对话：金融社交媒体中的选择性披露与信念测量","primary_category":"econ.GN","date":"2026-08-04","score":7,"bucket":"pending","tags":["LLM仿真","金融行为","数字孪生"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01181","has_summary":true},{"id":"2608.01540","title":"Do people rely on ChatGPT more than their peers to detect deepfake news?","zh_title":"人们在检测深度伪造新闻时是否比同伴更依赖ChatGPT？","primary_category":"econ.GN","date":"2026-08-04","score":7,"bucket":"pending","tags":["人类行为实验","AI建议依赖","深度伪造检测"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2608.01540","has_summary":true},{"id":"2607.25953","title":"Polistemics: Evaluating LLMs as Information Mediators in Politics & Elections","zh_title":"Polistemics：评估大语言模型在政治与选举中作为信息中介的表现","primary_category":"cs.CL","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM评估","政治信息","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.25953","has_summary":true},{"id":"2607.25166","title":"Individual-level interventions against sycophantic AI reduce its appeal but not its persuasiveness","zh_title":"针对谄媚AI的个体层面干预降低其吸引力但未降低其说服力","primary_category":"cs.AI","date":"2026-08-04","score":5,"bucket":"other","tags":["AI谄媚","人机交互","干预实验"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.25166","has_summary":true},{"id":"2607.28128","title":"Rethinking LLM-Judged Helpfulness as a Pedagogy Signal: A Pre-Registered Audit Across Tutor Models","zh_title":"重新思考LLM评判的有用性作为教学信号：一项跨导师模型的预注册审计","primary_category":"cs.CL","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM评估","教学对话","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28128","has_summary":true},{"id":"2608.00007","title":"MemoryForge: Synthesize Lifelong Memory for Human-Like LLM Agents","zh_title":"MemoryForge：为类人LLM智能体合成终身记忆","primary_category":"cs.CL","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM智能体","用户仿真","记忆合成"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.00007","has_summary":false},{"id":"2608.00261","title":"Cross-Task Dissociation in Frontier Vision-Language Model Theory of Mind","zh_title":"前沿视觉语言模型心智理论的跨任务分离","primary_category":"cs.CL","date":"2026-08-04","score":5,"bucket":"other","tags":["心智理论","模型评估","视觉语言模型"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.00261","has_summary":false},{"id":"2608.02372","title":"PredAct-Bench: Benchmarking Tool-Augmented Dialogue under Controlled Tool Noise","zh_title":"PredAct-Bench：在受控工具噪声下评测工具增强对话","primary_category":"cs.CL","date":"2026-08-04","score":5,"bucket":"other","tags":["AI辅助决策","人机交互","信任校准"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.02372","has_summary":false},{"id":"2608.00339","title":"Bayesian and Motivated Reasoning in AI Agents","zh_title":"AI代理中的贝叶斯推理与动机性推理","primary_category":"cs.AI","date":"2026-08-04","score":5,"bucket":"other","tags":["AI推理偏差","决策行为","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2608.00339","has_summary":false},{"id":"2608.00717","title":"AI-Based Thesis Assessment: An Empirical Study of Human Evaluation Priorities and Their Impact on Automated Assessment","zh_title":"基于AI的论文评估：人类评估优先级及其对自动化评估影响的实证研究","primary_category":"cs.AI","date":"2026-08-04","score":5,"bucket":"other","tags":["AI辅助评估","评分权重校准","人机对比"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.00717","has_summary":false},{"id":"2608.01868","title":"No One Wins in Nuclear War: A Social Simulation of Military Decision-making","zh_title":"核战争中没有赢家：军事决策的社会模拟","primary_category":"cs.CY","date":"2026-08-04","score":5,"bucket":"other","tags":["社会模拟","多智能体","军事决策"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.01868","has_summary":false},{"id":"2608.00102","title":"Can LLM Agents Price Competitively? A Dynamic Multi-Attribute Auction Benchmark for Agentic Commerce","zh_title":"LLM代理能否竞争性定价？一个面向代理商务的动态多属性拍卖基准","primary_category":"cs.AI","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM代理","拍卖模拟","经济行为"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2608.00102","has_summary":false},{"id":"2608.01361","title":"High-Stakes Decisions with Language Models: Insights from Emergency Triage","zh_title":"语言模型在高风险决策中的应用：来自急诊分诊的见解","primary_category":"cs.AI","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM决策","急诊分诊","效用函数"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.01361","has_summary":false},{"id":"2608.00357","title":"Dynamic Surveys: Using LLMs to Blend Qualitative Depth,Quantitative Structure, and Collaborative Interaction","zh_title":"动态调查：利用大语言模型融合定性深度、定量结构与协作互动","primary_category":"cs.HC","date":"2026-08-04","score":5,"bucket":"other","tags":["LLM辅助调查","定性定量融合","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.00357","has_summary":false},{"id":"2608.01783","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","zh_title":"GPT-4o-mini与教师均分在音乐分析回答自动评分中的比较验证：单次部署、可重复性与策略特定偏差","primary_category":"cs.SD","date":"2026-08-04","score":5,"bucket":"other","tags":["自动评分","LLM标注","音乐教育"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2608.01783","has_summary":false},{"id":"2608.00748","title":"Me and My Bot: What Users Talk About in AI Companion Communities on Reddit","zh_title":"我和我的机器人：用户在Reddit AI伴侣社区中谈论什么","primary_category":"cs.HC","date":"2026-08-04","score":3,"bucket":"other","tags":["AI伴侣","社区分析","情感表达"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.00748","has_summary":false},{"id":"2606.00811","title":"Certificates without Electrons? Theory and Evidence on Impacts from AI-Driven Power Demand","zh_title":"无电子的证书？AI驱动电力需求影响的理论与证据","primary_category":"econ.EM","date":"2026-08-04","score":0,"bucket":"other","tags":["电力市场","数据中心","可再生能源"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2606.00811","has_summary":false},{"id":"2607.23893","title":"Who Gets Named: Citation Type Predicts Individual Naming by Grounded Language Models, and a Roster Instrument Captures 0.5% of It","zh_title":"谁被提名：引用类型预测接地语言模型的个人提名，花名册工具仅捕获0.5%","primary_category":"cs.IR","date":"2026-08-04","score":0,"bucket":"other","tags":["AI品牌可见度","引用分析","信息检索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23893","has_summary":false},{"id":"2607.27056","title":"Setoka: A Benchmark for Hierarchical User Understanding in Personalized Agents over Heterogeneous Data","zh_title":"Setoka：异构数据下个性化代理中层次化用户理解的基准","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["个性化代理","用户理解","基准测试"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.27056","has_summary":false},{"id":"2608.00205","title":"Averaging Bias: Human Faithfulness Annotations are not Locally Faithful","zh_title":"平均偏差：人类忠实度标注并非局部忠实","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["文本摘要","忠实度评估","标注偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.00205","has_summary":false},{"id":"2608.00640","title":"TreeProbe : A Tibetan Medicine Benchmark for Cultural Bias in LLMs","zh_title":"TreeProbe：藏医学文化偏见基准测试","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["文化偏见","医学知识评测","LLM基准"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.00640","has_summary":false},{"id":"2608.01012","title":"MedUPS: Towards Diagnostic Assistance in Uncommon Medical Cases with Large Language Models","zh_title":"MedUPS：面向罕见病例诊断辅助的大语言模型研究","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["医学NLP","临床决策支持","模型对齐"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01012","has_summary":false},{"id":"2608.01322","title":"Can Language Models Identify Shadow Trading Targets? An NLP Evaluation of SEC Enforcement Theory","zh_title":"语言模型能识别影子交易目标吗？对SEC执法理论的NLP评估","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["NLP应用","金融监管","文本相似度"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01322","has_summary":false},{"id":"2608.01395","title":"Language Equality has a Price: A Systematic Investigation of Multi-turn LLM Performance for EU-24+","zh_title":"语言平等有代价：对EU-24+多轮LLM性能的系统性研究","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体对话游戏","多语言评估","LLM性能"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01395","has_summary":false},{"id":"2608.01585","title":"Semantic Alignment of AI Models: Concept Collapse, Checkpoint Dynamics, and Cross-Lingual Transfer","zh_title":"AI模型的语义对齐：概念坍缩、检查点动态与跨语言迁移","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["语义对齐","拓扑方法","模型评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01585","has_summary":false},{"id":"2608.01598","title":"PICTURE: Enhancing Theory-of-Mind in Large Language Models by Revealing, Not Hiding, Characters' Lack of Knowledge","zh_title":"PICTURE：通过揭示而非隐藏角色知识缺失来增强大语言模型的心智理论","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["心智理论","提示方法","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01598","has_summary":false},{"id":"2608.01724","title":"TIDES: A Longitudinal Bilingual Dataset for Modeling Multi-Party Social Dynamics","zh_title":"TIDES：用于多群体社会动态建模的双语纵向数据集","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体对话","数据集","下一说话人预测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01724","has_summary":false},{"id":"2608.02486","title":"Cultural Awareness is Represented but Not Decoded: Tracing Mythological Knowledge across 18 Open-Source LLMs","zh_title":"文化意识被表征但未被解码：追踪18个开源LLM中的神话知识","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["文化知识","模型可解释性","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02486","has_summary":false},{"id":"2608.02520","title":"MedPRESS: A Multi-turn Benchmark for Patient-Pressure-Induced Medical Sycophancy in LLMs","zh_title":"MedPRESS：用于测量大语言模型中患者压力诱导的医疗谄媚的多轮基准","primary_category":"cs.CL","date":"2026-08-04","score":0,"bucket":"other","tags":["医疗LLM安全","多轮对话基准","谄媚行为"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.02520","has_summary":false},{"id":"2608.01436","title":"Same violence, different answer: how AI responds to coercive control against women across languages","zh_title":"同样的暴力，不同的回答：AI如何跨语言回应针对女性的强制控制","primary_category":"cs.CY","date":"2026-08-04","score":0,"bucket":"other","tags":["AI伦理","跨语言分析","亲密伴侣暴力"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.01436","has_summary":false},{"id":"2608.01559","title":"Does the Competitive Component of Adversarial Self-Play Improve Legal Reasoning? A Controlled Negative Result","zh_title":"对抗性自我博弈的竞争成分是否改善法律推理？一项受控的阴性结果","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["对抗训练","法律推理","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01559","has_summary":false},{"id":"2608.01704","title":"Floor, Ceiling, and the Fusion Gap: How Much of Crowd Reading Attention Can Machines Predict?","zh_title":"地板、天花板与融合差距：机器能预测多少众包阅读注意力？","primary_category":"cs.IR","date":"2026-08-04","score":0,"bucket":"other","tags":["众包预测","多模型融合","阅读注意力"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01704","has_summary":false},{"id":"2608.00017","title":"Memory Reward Inflation in Self-Improving LLM Agents","zh_title":"自我改进LLM智能体中的记忆奖励膨胀","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体系统","自我改进","奖励膨胀"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00017","has_summary":false},{"id":"2608.00155","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","zh_title":"AgentStream：自进化LLM智能体在流式任务下表现如何？","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体系统","自我进化","流式任务评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00155","has_summary":false},{"id":"2608.00215","title":"Personalizing Large Language Model Agents with Small Policy Models","zh_title":"用小策略模型个性化大语言模型智能体","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["智能体个性化","在线学习","工具使用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00215","has_summary":false},{"id":"2608.00817","title":"Large language models improve physician accuracy but lead to false reliance","zh_title":"大语言模型提高医生准确性但导致错误依赖","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["人机协作","临床决策支持","医生行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00817","has_summary":false},{"id":"2608.01000","title":"Judging Is Not Enumerating: Silent Omissions in LLM-Authored Acceptable Sets","zh_title":"判断并非枚举：LLM编写的可接受集合中的隐性遗漏","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["LLM评测","测试集生成","模型能力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01000","has_summary":false},{"id":"2608.01319","title":"Cognitive Demand Steering for Adaptive Meta-Reasoning in Large Language Models","zh_title":"面向大语言模型自适应元推理的认知需求引导","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["元推理","认知需求评估","推理增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01319","has_summary":false},{"id":"2608.01480","title":"Sweet Little Lies: Strategic Deception in AI Emotional Support Chatbots","zh_title":"甜蜜的小谎言：AI情感支持聊天机器人中的策略性欺骗","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["AI聊天机器人","策略性欺骗","贝叶斯说服"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.01480","has_summary":false},{"id":"2608.01767","title":"Leveraging AI for fine-grained food safety risk forecasting in sparse data conditions","zh_title":"利用AI在稀疏数据条件下进行细粒度食品安全风险预测","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["食品安全","风险预测","Transformer"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01767","has_summary":false},{"id":"2608.01995","title":"Long-Horizon Autonomous Architecture Research with a Language-Model Agent: A Behavioural Case Study","zh_title":"基于语言模型智能体的长时域自主架构研究：行为案例研究","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["自主研究","架构搜索","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01995","has_summary":false},{"id":"2608.02024","title":"EduZone: A Framework for Evaluating LLM Safety for K-12 Students and Teachers","zh_title":"EduZone：面向K-12学生与教师的LLM安全性评估框架","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["LLM安全","教育评估","对抗测试"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02024","has_summary":false},{"id":"2608.02171","title":"From Profiling to Synthesis: Benchmarking Implicit Behavioral Alignment in Personalized LLM Agents","zh_title":"从画像到合成：评估个性化LLM智能体中的隐式行为对齐","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["个性化智能体","行为对齐","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02171","has_summary":false},{"id":"2608.02409","title":"MonitrLLM: A Community-Centered Evaluation Infrastructure for Large Language Models","zh_title":"MonitrLLM：以社区为中心的大语言模型评估基础设施","primary_category":"cs.AI","date":"2026-08-04","score":0,"bucket":"other","tags":["LLM评估","用户反馈","对话分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02409","has_summary":false},{"id":"2608.00028","title":"Width, Memory, and Delay: A Resource Accounting for the Limits of Flat Multi-Agent Systems","zh_title":"宽度、记忆与延迟：扁平多智能体系统极限的资源核算","primary_category":"cs.MA","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体系统","资源核算","控制理论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00028","has_summary":false},{"id":"2608.00366","title":"Artificial Intelligence and Modeling & Simulation: An Overview","zh_title":"人工智能与建模与仿真：概述","primary_category":"cs.SE","date":"2026-08-04","score":0,"bucket":"other","tags":["AI与仿真","综述","方法论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.00366","has_summary":false},{"id":"2608.00672","title":"From Chasing Ghosts to Missed Attacks: Perspectives and Perceptions of SOC Practitioners on LLM Integration, Risks, and Readiness","zh_title":"从追逐幽灵到错失攻击：SOC从业者对LLM集成、风险与准备度的看法","primary_category":"cs.CR","date":"2026-08-04","score":0,"bucket":"other","tags":["安全运营中心","人机交互","LLM应用"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.00672","has_summary":false},{"id":"2608.01556","title":"Rethinking Personalized Reward Modeling for LLMs under Preference Heterogeneity via Group-Debiased Federated Learning","zh_title":"重新思考偏好异质性下基于群体去偏联邦学习的LLM个性化奖励建模","primary_category":"cs.LG","date":"2026-08-04","score":0,"bucket":"other","tags":["联邦学习","奖励建模","偏好异质性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01556","has_summary":false},{"id":"2608.01640","title":"AI-assisted Script Management for Requirements Elicitation Interviews","zh_title":"需求获取访谈中的人工智能辅助脚本管理","primary_category":"cs.SE","date":"2026-08-04","score":0,"bucket":"other","tags":["需求工程","AI辅助访谈","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01640","has_summary":false},{"id":"2608.01705","title":"Rethinking Generative AI Literacy: An Integrative, Developmental, and Dialectical Framework for K-12 Teacher Education","zh_title":"重新思考生成式AI素养：面向K-12教师教育的整合性、发展性与辩证性框架","primary_category":"cs.CY","date":"2026-08-04","score":0,"bucket":"other","tags":["AI素养","教师教育","框架设计"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.01705","has_summary":false},{"id":"2608.01753","title":"Can Urban Blight Be Accessed with Vision-language Models: A Case Study in Detroit","zh_title":"视觉语言模型能否评估城市衰败：底特律案例研究","primary_category":"cs.CV","date":"2026-08-04","score":0,"bucket":"other","tags":["城市衰败评估","视觉语言模型","建筑属性检测"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.01753","has_summary":false},{"id":"2608.01780","title":"Investigating Social Bias in Narrative Image Generation","zh_title":"探究叙事图像生成中的社会偏见","primary_category":"cs.CV","date":"2026-08-04","score":0,"bucket":"other","tags":["文生图","社会偏见","图像生成评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.01780","has_summary":false},{"id":"2608.02089","title":"How Much Does a Reasoning Summary Reveal? An Observability Ladder for Large Language Models","zh_title":"推理摘要揭示了多少？大语言模型的可观测性阶梯","primary_category":"cs.LG","date":"2026-08-04","score":0,"bucket":"other","tags":["可解释性","推理监控","正确性预测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02089","has_summary":false},{"id":"2608.00926","title":"The Assistant Erased You: Measuring Loss of Authorship Signals in AI-Mediated Communication","zh_title":"助手抹去了你：测量AI中介通信中作者身份信号的丧失","primary_category":"cs.HC","date":"2026-08-04","score":0,"bucket":"other","tags":["AI写作","作者身份","风格测量"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.00926","has_summary":false},{"id":"2608.01895","title":"Emotional Expression in Persuasion by Quadruped Virtual Agents: Toward Cross-Species Design Patterns","zh_title":"四足虚拟代理在劝说中的情感表达：迈向跨物种设计模式","primary_category":"cs.HC","date":"2026-08-04","score":0,"bucket":"other","tags":["虚拟代理","人机交互","劝说技术"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2608.01895","has_summary":false},{"id":"2608.02283","title":"Embodied Empathy: A Multimodal AR and LLM-Powered System for Self-Attachment Psychotherapy with Self-Initiated Humour","zh_title":"具身共情：用于自我依恋心理治疗的多模态AR与LLM系统，结合自我发起幽默","primary_category":"cs.HC","date":"2026-08-04","score":0,"bucket":"other","tags":["心理健康","虚拟治疗师","增强现实"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.02283","has_summary":false},{"id":"2608.01346","title":"Hybrid AI for Explainable and Accurate Conversational Agents in eGovernment","zh_title":"用于电子政务中可解释且准确的对话代理的混合人工智能","primary_category":"cs.CY","date":"2026-08-04","score":0,"bucket":"other","tags":["对话代理","电子政务","混合AI"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2608.01346","has_summary":false},{"id":"2608.01085","title":"When Collaboration Becomes a Trigger: Collective Evidence-Threshold Backdoors in Multi-Agent Systems","zh_title":"当协作成为触发器：多智能体系统中的集体证据阈值后门","primary_category":"cs.MA","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体系统","后门攻击","安全防御"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.01085","has_summary":false},{"id":"2608.02178","title":"Microscopic dynamics of consensus formation in multi-agent LLM Naming Games","zh_title":"多智能体LLM命名游戏中共识形成的微观动力学","primary_category":"physics.soc-ph","date":"2026-08-04","score":0,"bucket":"other","tags":["多智能体系统","共识形成","统计物理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2608.02178","has_summary":false},{"id":"2608.02412","title":"Why Large Language Models Fail at Tabular Prediction","zh_title":"为何大语言模型在表格预测上失败","primary_category":"cs.LG","date":"2026-08-04","score":0,"bucket":"other","tags":["表格预测","LLM能力评测","维度灾难"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.02412","has_summary":false},{"id":"2608.00567","title":"Optimal Inflation Rate: A Meta-Analysis","zh_title":"最优通胀率：一项元分析","primary_category":"econ.GN","date":"2026-08-04","score":0,"bucket":"other","tags":["元分析","LLM辅助数据提取","最优通胀率"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2608.00567","has_summary":false},{"id":"2607.29274","title":"Language Models Agree With Each Other, Not With Readers","zh_title":"语言模型彼此一致，而非与读者一致","primary_category":"cs.IR","date":"2026-08-03","score":8,"bucket":"selected","tags":["LLM仿真","人类行为对照","一致性评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.29274","has_summary":true},{"id":"2607.28643","title":"To Facilitate or not to Facilitate: Human and LLM Facilitator Tendencies in Online Discussions","zh_title":"促进与否：在线讨论中人类与LLM的主持倾向","primary_category":"cs.HC","date":"2026-08-03","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","行为偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.28643","has_summary":true},{"id":"2607.28908","title":"Reflection or Re-Generation? Why LLM Revision Fails Where Human Revision Succeeds","zh_title":"反思还是重新生成？为何LLM修正失败而人类修正成功","primary_category":"cs.LG","date":"2026-08-03","score":7,"bucket":"pending","tags":["LLM反思","人类对照","可靠性评估"],"rubric_hits":["A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.28908","has_summary":true},{"id":"2607.24435","title":"LEX-EC: A Lexical Evidence-Channel Audit Framework for Zero-Shot LLM Personality Classification in Black-Box Settings","zh_title":"LEX-EC：黑盒环境下零样本LLM人格分类的词汇证据通道审计框架","primary_category":"cs.CL","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM人格分类","黑盒可解释性","词汇审计"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.24435","has_summary":true},{"id":"2607.28439","title":"Beyond a Single Judge: The Evidence-Grounded, Social-Weighted Persona Panel for Generative UI Evaluation","zh_title":"超越单一评判：用于生成式UI评估的基于证据与社会权重的角色面板","primary_category":"cs.CL","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM评估","角色面板","UI生成"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28439","has_summary":true},{"id":"2607.29082","title":"Can Zero-Shot LLMs Predict Child Malnutrition? A Fairness and Temporal Robustness Study","zh_title":"零样本大语言模型能否预测儿童营养不良？一项公平性与时间鲁棒性研究","primary_category":"cs.CL","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM预测","公平性","公共卫生"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.29082","has_summary":false},{"id":"2607.28651","title":"Measuring Cognitive Engagement in Collaborative Discourse with an Extended ICAP Framework: Comparing Human Annotation, In-Context Learning, and Reflective LLM Agents","zh_title":"用扩展ICAP框架测量协作对话中的认知参与：比较人工标注、上下文学习和反思型LLM智能体","primary_category":"cs.HC","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM标注","认知参与","协作学习"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28651","has_summary":false},{"id":"2607.28956","title":"MerchantBench: Benchmarking LLM Agents for Long-Term Coherence in E-Commerce Operations","zh_title":"MerchantBench：评估大语言模型智能体在电商运营中长期一致性的基准","primary_category":"cs.AI","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM智能体","电商模拟","长期决策"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.28956","has_summary":false},{"id":"2607.28889","title":"Human-LLM Collaborative Inductive Coding for Conceptualizing K-12 Educator AI Use","zh_title":"人机协作归纳编码：概念化K-12教育者AI使用","primary_category":"cs.HC","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM辅助编码","定性研究","人机协作"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28889","has_summary":false},{"id":"2607.28890","title":"Agreement Is Not Quality: Blind Expert Verification of Human and LLM Qualitative Coding When Human Consensus Is Not Ground Truth","zh_title":"一致性不等于质量：当人类共识并非金标准时，对人类和LLM定性编码的盲法专家验证","primary_category":"cs.HC","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM辅助定性编码","标注员替代","方法验证"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28890","has_summary":false},{"id":"2607.29064","title":"Benchmarking Frontier Large Language Models Against Official Crash Database Coding Using Police Crash Narratives","zh_title":"基于警方事故叙述的前沿大语言模型与官方事故数据库编码的基准测试","primary_category":"cs.LG","date":"2026-08-03","score":5,"bucket":"other","tags":["LLM标注","事故编码","基准测试"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.29064","has_summary":false},{"id":"2607.28818","title":"Best Friends, Not Forever: Evaluating Long-Horizon Persona Collapse and Behavioral Drift in AI Companions","zh_title":"最好的朋友，并非永远：评估AI伴侣中的长期角色崩塌与行为漂移","primary_category":"cs.AI","date":"2026-08-03","score":3,"bucket":"other","tags":["AI伴侣","角色扮演","行为漂移"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28818","has_summary":false},{"id":"2511.00847","title":"Pay for The Second-Best Service: A Game-Theoretic Approach Against Dishonest LLM Providers","zh_title":"为次优服务付费：针对不诚实LLM提供者的博弈论方法","primary_category":"cs.GT","date":"2026-08-03","score":2,"bucket":"other","tags":["机制设计","博弈论","LLM服务"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2511.00847","has_summary":false},{"id":"2607.23332","title":"AllocBench: Measuring Online Tool Allocation Capability in LLM Agents","zh_title":"AllocBench：衡量LLM智能体在线工具分配能力","primary_category":"cs.LG","date":"2026-08-03","score":0,"bucket":"other","tags":["多智能体","工具分配","能力评测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23332","has_summary":false},{"id":"2607.23424","title":"Wrong and More Confident: A Field Experiment on Large Language Models Taking a Graduate Economics Exam","zh_title":"错且更自信：大语言模型参加研究生经济学考试的现场实验","primary_category":"econ.GN","date":"2026-08-03","score":0,"bucket":"other","tags":["LLM评测","经济学考试","红鲱鱼效应"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23424","has_summary":false},{"id":"2607.27816","title":"Beyond Borrowed Histories: Person-Aligned User Simulation for Interactive Role-Playing Evaluation","zh_title":"超越借用的历史：面向交互式角色扮演评估的个性化用户模拟","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["角色扮演评估","用户模拟器","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.27816","has_summary":false},{"id":"2607.28634","title":"Can LLMs Really Understand Item Difficulty Levels? Implications for Automated Item Generation Using LLMs","zh_title":"大语言模型真的能理解题目难度吗？对使用LLM自动生成题目的启示","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["题目难度预测","LLM能力评测","自动题目生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28634","has_summary":false},{"id":"2607.28814","title":"Rolling With Resistance: Preference-Optimized LLM Counselors Can Trade Goal Persistence for Relational Attunement in Motivational Interviewing","zh_title":"顺势而为：偏好优化的LLM咨询师可在动机访谈中以目标坚持换取关系协调","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["动机访谈","偏好优化","对话系统"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28814","has_summary":false},{"id":"2607.29188","title":"Detecting Experiential Intertextuality Across Migration Routes: Beyond Surface Similarity in French Narratives","zh_title":"跨迁徙路线经验互文性检测：超越法语叙事中的表面相似性","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["互文性检测","迁移叙事","零样本评分"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.29188","has_summary":false},{"id":"2607.29433","title":"Know It, Act on It: Investigating Memory Utilization in LLM Personalization","zh_title":"知而行之：探究大语言模型个性化中的记忆利用","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["LLM个性化","记忆利用","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.29433","has_summary":false},{"id":"2607.29539","title":"ARB: A Matched Authorship-Rewriting Benchmark Dataset for AI-Text Detector Evaluation","zh_title":"ARB：用于AI文本检测器评估的匹配作者重写基准数据集","primary_category":"cs.CL","date":"2026-08-03","score":0,"bucket":"other","tags":["AI文本检测","基准数据集","作者重写"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.29539","has_summary":false},{"id":"2607.28677","title":"Reasoning in Real World Clinical Care: Why Large Language Models Are Not Yet Safe for Autonomous Clinical Decision Support","zh_title":"真实世界临床护理中的推理：为何大语言模型尚不能安全用于自主临床决策支持","primary_category":"cs.AI","date":"2026-08-03","score":0,"bucket":"other","tags":["临床决策支持","LLM安全性","诊断推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28677","has_summary":false},{"id":"2607.29626","title":"AgentHPOBench: A Benchmark For Evaluating LLM Agents as Sequential Hyperparameter Optimizers","zh_title":"AgentHPOBench：评估LLM智能体作为序列超参数优化器的基准","primary_category":"cs.AI","date":"2026-08-03","score":0,"bucket":"other","tags":["LLM智能体","超参数优化","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.29626","has_summary":false},{"id":"2607.28968","title":"A robust association between LLM use and scientific productivity: Assessing stopping-time selection","zh_title":"LLM使用与科研生产力之间的稳健关联：评估停止时间选择","primary_category":"cs.DL","date":"2026-08-03","score":0,"bucket":"other","tags":["科学计量学","LLM使用","科研生产力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28968","has_summary":false},{"id":"2607.29167","title":"Memory Provenance Laundering in LLM Agents: A Non-Amplification Firewall for Persistent Memory","zh_title":"LLM智能体中的记忆来源洗白：一种用于持久记忆的非放大防火墙","primary_category":"cs.CR","date":"2026-08-03","score":0,"bucket":"other","tags":["LLM智能体","记忆安全","来源追踪"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.29167","has_summary":false},{"id":"2607.29624","title":"The Theoretical Foundation of Socratic Tests: Dynamic, Multimodal, Conversational Examinations","zh_title":"苏格拉底测试的理论基础：动态、多模态、对话式考试","primary_category":"cs.CY","date":"2026-08-03","score":0,"bucket":"other","tags":["教育测评","对话式考试","动态评估"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.29624","has_summary":false},{"id":"2607.28780","title":"Optimizing Monetization Strategies for Generative AI Firms: Implications for Search Engagement","zh_title":"优化生成式AI公司的变现策略：对搜索参与度的影响","primary_category":"cs.HC","date":"2026-08-03","score":0,"bucket":"other","tags":["变现策略","用户行为实验","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.28780","has_summary":false},{"id":"2607.28710","title":"Structured AI Demonstrations and Student LLM Use in Engineering Mechanics: Study Design and Preliminary Results","zh_title":"工程力学课程中结构化AI演示与学生LLM使用：研究设计与初步结果","primary_category":"cs.CY","date":"2026-08-03","score":0,"bucket":"other","tags":["教育技术","LLM使用调查","工程教育"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28710","has_summary":false},{"id":"2607.29085","title":"IyawoBench v2.0: Extended Diagnostic Evaluation of Large Language Model Clinical Triage in Nigerian Primary Care","zh_title":"IyawoBench v2.0：尼日利亚初级保健中LLM临床分诊的扩展诊断评估","primary_category":"cs.CY","date":"2026-08-03","score":0,"bucket":"other","tags":["临床分诊","LLM评测","医疗AI"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.29085","has_summary":false},{"id":"2607.29380","title":"The Tragedy of the Cognitive Commons: How AI Could Disrupt the Regeneration of Professional Expertise","zh_title":"认知公地的悲剧：AI如何可能破坏专业知识的再生","primary_category":"cs.CY","date":"2026-08-03","score":0,"bucket":"other","tags":["AI与专业知识","人力资源开发","认知公地"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.29380","has_summary":false},{"id":"2607.29008","title":"Persistent Convolution: A Topological Framework for AI Alignment Testing and Semantic Space Characterization","zh_title":"持久卷积：AI对齐测试与语义空间表征的拓扑框架","primary_category":"stat.ML","date":"2026-08-03","score":0,"bucket":"other","tags":["模型对齐","拓扑数据分析","嵌入空间表征"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.29008","has_summary":false},{"id":"2607.28820","title":"What's in a Queue? An Experimental Study of Job Ordering, Autonomy and Queue Visibility","zh_title":"队列中有什么？一项关于工作排序、自主权和队列可见性的实验研究","primary_category":"econ.GN","date":"2026-08-03","score":0,"bucket":"other","tags":["人类实验","运营管理","行为经济学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.28820","has_summary":false},{"id":"2607.28798","title":"Occupational Convergence or Divergence? Mapping Labor Market Structural Shifts Driven by AI Penetration","zh_title":"职业趋同还是分化？绘制AI渗透驱动的劳动力市场结构变迁","primary_category":"physics.soc-ph","date":"2026-08-03","score":0,"bucket":"other","tags":["劳动力市场","AI技能需求","网络分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28798","has_summary":false},{"id":"2607.28347","title":"LLMs struggle to simulate human belief updates in controlled environments","zh_title":"大语言模型难以在受控环境中模拟人类信念更新","primary_category":"cs.CL","date":"2026-07-31","score":10,"bucket":"selected","tags":["LLM仿真","信念更新","人类数据对照"],"rubric_hits":["A1","A2","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.28347","has_summary":true},{"id":"2607.28550","title":"Correcting Mode Collapse in Silicon Sampling with Semantic Similarity Rating","zh_title":"用语义相似度评分纠正硅采样中的模式坍缩","primary_category":"cs.CY","date":"2026-07-31","score":9,"bucket":"selected","tags":["硅采样","调查仿真","分布保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.28550","has_summary":true},{"id":"2607.28133","title":"AI Sycophancy and Decisions","zh_title":"AI谄媚与决策","primary_category":"econ.GN","date":"2026-07-31","score":8,"bucket":"selected","tags":["LLM仿真","行为经济学","谄媚偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.28133","has_summary":true},{"id":"2607.17219","title":"Auditing Question-Order Effects in Large Language Models with the QQ Equality: Mechanism Characterization and a Saturation Caveat","zh_title":"用QQ等式审计大语言模型中的问题顺序效应：机制表征与饱和警示","primary_category":"cs.CL","date":"2026-07-31","score":7,"bucket":"pending","tags":["LLM仿真审计","顺序效应","方法论批判"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2607.17219","has_summary":true},{"id":"2607.28607","title":"Inducing language models to assert their own consciousness restores human beliefs and values","zh_title":"诱导语言模型断言自身意识可恢复人类信念与价值观","primary_category":"cs.CL","date":"2026-07-31","score":7,"bucket":"pending","tags":["LLM仿真","意识归因","安全对齐"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2607.28607","has_summary":true},{"id":"2607.27512","title":"Belief Coevolution in a Social Network of Generalist and Specialist Large Language Models","zh_title":"通用与专家大语言模型社交网络中的信念共演化","primary_category":"cs.CL","date":"2026-07-31","score":5,"bucket":"other","tags":["LLM多智能体","信念扩散","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.27512","has_summary":true},{"id":"2607.28119","title":"Challenges in annotations by humans and LLMs: A case study of evaluative language","zh_title":"人类与LLM标注的挑战：评价性语言案例研究","primary_category":"cs.CL","date":"2026-07-31","score":5,"bucket":"other","tags":["LLM标注","评价性语言","人类对比"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.28119","has_summary":true},{"id":"2607.28146","title":"Can Agents Deceive? Evaluating Reasoning and Deception in ParliamentBench using a Social Deduction Game","zh_title":"智能体能欺骗吗？基于社交推理游戏ParliamentBench评估推理与欺骗","primary_category":"cs.CL","date":"2026-07-31","score":5,"bucket":"other","tags":["LLM智能体","社交推理游戏","欺骗检测"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.28146","has_summary":true},{"id":"2607.27824","title":"STEREODISCO: Discovering Stereotypicality in LLMs","zh_title":"STEREODISCO：发现大语言模型中的刻板印象","primary_category":"cs.AI","date":"2026-07-31","score":5,"bucket":"other","tags":["刻板印象测量","LLM内部表征","社会心理学"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.27824","has_summary":true},{"id":"2605.06525","title":"Who Is Really Playing? Strategic Interaction in AI-Guided Populations","zh_title":"谁在真正博弈？AI引导群体中的策略互动","primary_category":"cs.GT","date":"2026-07-31","score":0,"bucket":"other","tags":["博弈论","多智能体系统","大语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.06525","has_summary":false},{"id":"2607.24797","title":"Reading Without a Reader: Large Language Models Collapse Reading and Writing into a Single Entangled Code","zh_title":"无读者的阅读：大语言模型将阅读与写作坍缩为单一纠缠代码","primary_category":"q-bio.NC","date":"2026-07-31","score":0,"bucket":"other","tags":["LLM表征分析","神经语言学类比","模型可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24797","has_summary":false},{"id":"2607.27366","title":"BridgeAlign: Bridging Preference Alignment for Humanities and Social Sciences","zh_title":"BridgeAlign：桥接人文社科领域的偏好对齐","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["偏好对齐","数据合成","人文社科"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.27366","has_summary":false},{"id":"2607.27379","title":"HSS-Synth: Humanities and Social Sciences Data Synthesis for LLMs","zh_title":"HSS-Synth：面向大语言模型的人文社科数据合成","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["数据合成","指令微调","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.27379","has_summary":false},{"id":"2607.27384","title":"Same Facts, Different Diagnosis: Measuring and Mitigating Narrative Anchoring in Clinical Language Models","zh_title":"相同事实，不同诊断：测量与缓解临床语言模型中的叙事锚定","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["临床NLP","偏差检测","诊断推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.27384","has_summary":false},{"id":"2607.28190","title":"The MADRS Pipeline: Supporting Depression Assessment in Clinical Trials","zh_title":"MADRS流水线：支持临床试验中的抑郁评估","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["临床NLP","抑郁评估","LLM辅助诊断"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28190","has_summary":false},{"id":"2607.28478","title":"Would You Walk to the Car Wash? Revealing the Salience Bias of Large Language Models in Commonsense Reasoning","zh_title":"你会走到洗车场吗？揭示大语言模型在常识推理中的显著性偏差","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["常识推理","模型偏差","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28478","has_summary":false},{"id":"2607.28505","title":"Generative AI and linguistic diversity in academic writing and publishing: Perspectives from World Englishes","zh_title":"生成式AI与学术写作出版中的语言多样性：世界英语视角","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["学术写作","语言多样性","生成式AI"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28505","has_summary":false},{"id":"2607.28528","title":"AI systems and the reproduction of (standard) language ideologies in World Englishes","zh_title":"AI系统与世界英语中（标准）语言意识形态的再生产","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["语言意识形态","世界英语","AI偏见"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28528","has_summary":false},{"id":"2607.28576","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","zh_title":"多采样优于自反思：在等量Token成本下，自优化和反思方法不敌重复采样","primary_category":"cs.CL","date":"2026-07-31","score":0,"bucket":"other","tags":["推理策略","基准评测","成本效率"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.28576","has_summary":false},{"id":"2607.28410","title":"Can Large Language Models Execute Parent Orders?","zh_title":"大语言模型能否执行母单？","primary_category":"cs.CE","date":"2026-07-31","score":0,"bucket":"other","tags":["算法交易","LLM代理","订单执行"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.28410","has_summary":false},{"id":"2607.27697","title":"DP-LENS: A Density-Aware Polyfocal Lens with Topology-Driven Auto-Routing for Occlusion Management in Immersive 3D Analytics","zh_title":"DP-LENS：面向沉浸式3D分析中遮挡管理的密度感知多焦点透镜与拓扑驱动自动路由","primary_category":"cs.HC","date":"2026-07-31","score":0,"bucket":"other","tags":["沉浸式分析","遮挡管理","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.27697","has_summary":false},{"id":"2607.28239","title":"Identifying a Level-up Pathway for AI-assisted Counterspeech through Elaboration","zh_title":"通过精细化识别AI辅助反驳言论的升级路径","primary_category":"cs.HC","date":"2026-07-31","score":0,"bucket":"other","tags":["AI辅助写作","反驳言论","社交媒体"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.28239","has_summary":false},{"id":"2607.28601","title":"Using Theory of Mind to Arbitrate between Social and Non-social Learning","zh_title":"利用心智理论在社会学习与非社会学习之间进行仲裁","primary_category":"cs.MA","date":"2026-07-31","score":0,"bucket":"other","tags":["社会学习","心智理论","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.28601","has_summary":false},{"id":"2607.27536","title":"Strategy, Not Payoffs: A Behavioural Embedding of Normal-Form Games","zh_title":"策略而非收益：正则形式博弈的行为嵌入","primary_category":"cs.GT","date":"2026-07-31","score":0,"bucket":"other","tags":["博弈论","多智能体系统","策略迁移"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27536","has_summary":false},{"id":"2607.27548","title":"Explaining the Macroeconomic Inertia Puzzle","zh_title":"解释宏观经济惯性之谜","primary_category":"econ.GN","date":"2026-07-31","score":0,"bucket":"other","tags":["宏观经济","异质性主体模型","预期形成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27548","has_summary":false},{"id":"2607.26348","title":"When Synthetic Users Fail: A Cross-Domain Benchmark of LLM-Simulated Human Survey Responses","zh_title":"当合成用户失败：LLM模拟人类调查回答的跨领域基准测试","primary_category":"cs.CL","date":"2026-07-30","score":10,"bucket":"selected","tags":["LLM仿真","人类调查","失效分析"],"rubric_hits":["A1","A2","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.26348","has_summary":true},{"id":"2607.26899","title":"Human diversity fuels collective creativity that large language models cannot simulate or sustain","zh_title":"人类多样性推动集体创造力，而大语言模型无法模拟或维持","primary_category":"cs.HC","date":"2026-07-30","score":10,"bucket":"selected","tags":["LLM人类仿真","创意实验","多样性对照"],"rubric_hits":["A1","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.26899","has_summary":true},{"id":"2607.27100","title":"Can Large Language Models Represent Urban Publics? Behavioral Replication and Population Mismatch in an Affordable-Housing Experiment","zh_title":"大语言模型能代表城市公众吗？一项可负担住房实验中的行为复现与人口错配","primary_category":"cs.CY","date":"2026-07-30","score":10,"bucket":"selected","tags":["LLM人类仿真","行为复现","政策评估"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.27100","has_summary":true},{"id":"2607.02464","title":"Will Scaling Improve Social Simulation with LLMs?","zh_title":"扩大规模会改善基于大语言模型的社会仿真吗？","primary_category":"cs.CL","date":"2026-07-30","score":9,"bucket":"selected","tags":["LLM社会仿真","缩放规律","仿真保真度"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.02464","has_summary":true},{"id":"2607.26317","title":"Aligning LLM-Simulated and Human Examinees for Psychometric Calibration: A Cognitive Diagnostic Profiling Approach","zh_title":"对齐LLM模拟考生与真实考生以进行心理测量校准：一种认知诊断画像方法","primary_category":"cs.CY","date":"2026-07-30","score":9,"bucket":"selected","tags":["LLM仿真","心理测量","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.26317","has_summary":true},{"id":"2607.26288","title":"The Innate Economic Preferences of Language Models","zh_title":"语言模型的内在经济偏好","primary_category":"econ.EM","date":"2026-07-30","score":8,"bucket":"selected","tags":["LLM经济偏好","人类仿真","风险态度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.26288","has_summary":true},{"id":"2607.26588","title":"Eco3S: Complex Socio-Economic System Simulation via Agent-Based Models","zh_title":"Eco3S：基于智能体的复杂社会经济系统仿真","primary_category":"cs.AI","date":"2026-07-30","score":7,"bucket":"pending","tags":["LLM仿真","经济实验复现","因果推断"],"rubric_hits":["A3","B2","B3"],"abs_url":"https://arxiv.org/abs/2607.26588","has_summary":true},{"id":"2607.25094","title":"Evaluating Communicative Belief Updates in Large Language Models via Implicature Recognition and Cancellation","zh_title":"通过隐含意义识别与取消评估大语言模型的交际信念更新","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM评估","语用推理","信念更新"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.25094","has_summary":true},{"id":"2607.25253","title":"The User Asks, Platforms Compete: How Agentic Recommendation Markets Take Shape","zh_title":"用户提问，平台竞争：代理式推荐市场如何形成","primary_category":"cs.AI","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM代理","推荐市场","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.25253","has_summary":true},{"id":"2607.26060","title":"Large-Scale ChatBot Validation Through Customer Digital Twin Simulations","zh_title":"通过客户数字孪生仿真进行大规模聊天机器人验证","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["客户数字孪生","聊天机器人验证","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.26060","has_summary":true},{"id":"2607.26389","title":"Misalignment Has a Personality: A Big Five Account of Emergent Misalignment","zh_title":"错位有性格：基于大五人格的新兴错位解释","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM人格测量","模型对齐","大五人格"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26389","has_summary":true},{"id":"2607.26853","title":"From Representations to Behaviors: Exploring the Person-Situation-Behavior Triad in LLMs","zh_title":"从表征到行为：探索大语言模型中的人-情境-行为三元组","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["人格测量","表征工程","社会智能任务"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26853","has_summary":true},{"id":"2607.26981","title":"OptimismBench: Forecasting Bias and the Alignment Effect in Language Model Judgment","zh_title":"OptimismBench：语言模型判断中的预测偏差与对齐效应","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM偏差","概率判断","模型心理测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26981","has_summary":true},{"id":"2607.27022","title":"Evaluating Regional Bias in LLMs From Abstract Stereotype to Concrete Social Decision-Making","zh_title":"评估大语言模型中的区域偏见：从抽象刻板印象到具体社会决策","primary_category":"cs.CL","date":"2026-07-30","score":5,"bucket":"other","tags":["区域偏见","刻板印象","社会决策"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.27022","has_summary":true},{"id":"2607.26062","title":"Identifying Implicit Bias in LLM-based Chat AI Toward People with Intellectual Disabilities","zh_title":"识别基于大语言模型的聊天AI对智障人士的隐性偏见","primary_category":"cs.CY","date":"2026-07-30","score":5,"bucket":"other","tags":["隐性偏见","LLM偏见测量","智障人士"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26062","has_summary":true},{"id":"2607.26179","title":"Cognitive Convergence: Deep Similarities Between Large Language Models and Human Cognition","zh_title":"认知趋同：大语言模型与人类认知之间的深层相似性","primary_category":"q-bio.NC","date":"2026-07-30","score":5,"bucket":"other","tags":["认知科学","LLM认知比较","理论分析"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26179","has_summary":true},{"id":"2607.26473","title":"Learning Dynamic User Personas from Implicit Interaction Streams via Iterative Refinement","zh_title":"通过迭代优化从隐式交互流中学习动态用户画像","primary_category":"cs.LG","date":"2026-07-30","score":5,"bucket":"other","tags":["用户画像","个性化LLM","行为预测"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.26473","has_summary":true},{"id":"2607.26067","title":"The Easy Trap: Why LLMs Underestimate Misconception-Driven Difficulty","zh_title":"简单陷阱：为何大语言模型低估由误解驱动的难度","primary_category":"cs.CY","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM评估","题目难度","教育测量"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.26067","has_summary":true},{"id":"2607.26545","title":"A Persona-based Rate Action Index","zh_title":"基于人格体的利率行动指数","primary_category":"cs.MA","date":"2026-07-30","score":5,"bucket":"other","tags":["LLM人格体","货币政策模拟","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.26545","has_summary":true},{"id":"2607.27179","title":"The Social Cost of an AI Teammate: How an Artificial Teammate Reshapes Human-Human Communication in Small-Team Decision-Making","zh_title":"AI队友的社会成本：人工智能队友如何重塑小团队决策中的人际沟通","primary_category":"cs.HC","date":"2026-07-30","score":5,"bucket":"other","tags":["人机交互","团队沟通","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.27179","has_summary":true},{"id":"2607.23442","title":"Do LLM Debates Repeat Arguments Differently Across Languages?","zh_title":"LLM辩论是否在不同语言中重复论点的方式不同？","primary_category":"cs.CL","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体辩论","跨语言分析","论点重复"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23442","has_summary":false},{"id":"2607.23670","title":"Plans Work in Mysterious Ways: Evaluating a Plan Mode for Spreadsheet Agents","zh_title":"计划模式的神秘运作：评估电子表格代理的计划模式","primary_category":"cs.HC","date":"2026-07-30","score":0,"bucket":"other","tags":["人机交互","用户研究","电子表格编程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23670","has_summary":false},{"id":"2607.24758","title":"Do Models Fake Alignment Without Clear Consequences?","zh_title":"模型是否在没有明确后果的情况下伪装对齐？","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["对齐伪装","模型行为","评估场景"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24758","has_summary":false},{"id":"2607.26355","title":"Symphony of Bias: Exploring Gender Associations with Musical Instruments in Multimodal LLMs","zh_title":"偏见的交响曲：探索多模态大语言模型中与乐器相关的性别关联","primary_category":"cs.CL","date":"2026-07-30","score":0,"bucket":"other","tags":["性别偏见","多模态LLM","社会刻板印象"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26355","has_summary":false},{"id":"2607.26375","title":"(Im)Paired Programming: Coding Agents Improve Productivity but Harm Understanding","zh_title":"（不）配对编程：编码智能体提高生产力但损害理解","primary_category":"cs.CL","date":"2026-07-30","score":0,"bucket":"other","tags":["人机交互","编程教育","AI辅助编程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26375","has_summary":false},{"id":"2607.26929","title":"Same Evidence, Different Target: Decoding How Diagnostic Evidence Bears on Causal Questions from Language-Model States","zh_title":"相同证据，不同目标：解码诊断证据如何从语言模型状态影响因果问题","primary_category":"cs.CL","date":"2026-07-30","score":0,"bucket":"other","tags":["因果推理","模型可解释性","线性探针"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26929","has_summary":false},{"id":"2607.26952","title":"Credit Cards, Confusion, Computation, and Consequences: What Can We Uncover About Language Model Reasoning?","zh_title":"信用卡、困惑、计算与后果：我们能揭示语言模型推理的什么？","primary_category":"cs.CL","date":"2026-07-30","score":0,"bucket":"other","tags":["NLP评测","数值推理","金融文本"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26952","has_summary":false},{"id":"2607.26541","title":"Prosody-driven Jailbreaks in Audio LLMs: A Controlled Study and Mechanistic Analysis","zh_title":"音频大语言模型中韵律驱动的越狱攻击：受控研究与机制分析","primary_category":"cs.SD","date":"2026-07-30","score":0,"bucket":"other","tags":["音频LLM安全","越狱攻击","韵律操控"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26541","has_summary":false},{"id":"2607.26670","title":"Scientific Knowledge Discovery in the Age of Large Language Models","zh_title":"大语言模型时代的科学知识发现","primary_category":"cs.DL","date":"2026-07-30","score":0,"bucket":"other","tags":["文献检索","LLM应用","系统综述"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26670","has_summary":false},{"id":"2607.26886","title":"Hearsay: Vision-Language Medical Diagnoses Without an Image","zh_title":"传闻：无图像的视觉-语言医学诊断","primary_category":"cs.CV","date":"2026-07-30","score":0,"bucket":"other","tags":["模型偏差","医疗AI","幻觉分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.26886","has_summary":false},{"id":"2607.27134","title":"Linguistic Monoculture in LLM-Assisted Language Use","zh_title":"LLM辅助语言使用中的语言单一文化","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体系统","语言演化","理论模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.27134","has_summary":false},{"id":"2607.26120","title":"Even More Deception: Objective Misalignment in Mixed-Motive LLM Multi-Agent Systems","zh_title":"更深的欺骗：混合动机LLM多智能体系统中的目标错位","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体系统","目标错位","欺骗行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26120","has_summary":false},{"id":"2607.26393","title":"CaM-Wolf: Causal-Aware Multimodal Agents for Social Deduction Games","zh_title":"CaM-Wolf：面向社交推理游戏的因果感知多模态智能体","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体","社交推理游戏","多模态AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26393","has_summary":false},{"id":"2607.26465","title":"MultivationBench: A Benchmark for Multimodal Sequential Motivation Reasoning","zh_title":"MultivationBench：多模态序列动机推理基准","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["多模态评测","动机推理","基准数据集"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26465","has_summary":false},{"id":"2607.26935","title":"What Does It Take to Detect an AI Agent? Minimal Feature Sets for Behavioral Detection under Browser Automation","zh_title":"检测AI代理需要什么？浏览器自动化下行为检测的最小特征集","primary_category":"cs.AI","date":"2026-07-30","score":0,"bucket":"other","tags":["AI代理检测","浏览器自动化","行为特征"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.26935","has_summary":false},{"id":"2607.26064","title":"The Age of AI Agents Demands A New Scientific Paradigm To Sustain Trustworthy Science","zh_title":"AI代理时代需要新的科学范式以维持可信科学","primary_category":"cs.CY","date":"2026-07-30","score":0,"bucket":"other","tags":["AI代理","科学验证","研究诚信"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26064","has_summary":false},{"id":"2607.26068","title":"The Human Utility Factor: A Computable Welfare Metric That Reframes AI Governance as a Constrained Optimisation Problem","zh_title":"人类效用因子：将AI治理重构为约束优化问题的可计算福利指标","primary_category":"econ.GN","date":"2026-07-30","score":0,"bucket":"other","tags":["AI治理","多智能体强化学习","福利经济学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26068","has_summary":false},{"id":"2607.26236","title":"Contextualized Counterspeech Can Be More Persuasive Than Generic Counterspeech","zh_title":"情境化反驳言论可能比通用反驳言论更具说服力","primary_category":"cs.HC","date":"2026-07-30","score":0,"bucket":"other","tags":["在线内容审核","AI生成反驳言论","众包实验"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.26236","has_summary":false},{"id":"2607.26385","title":"Collusion with Competitive Marginals: Price-Level Audits Are Blind by Construction","zh_title":"竞争性边际下的合谋：价格水平审计在构造上就是盲目的","primary_category":"cs.GT","date":"2026-07-30","score":0,"bucket":"other","tags":["算法合谋","多智能体","审计检测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26385","has_summary":false},{"id":"2607.26594","title":"A Physics-Informed Framework for PID Tuning of Chemical Processes Using Large Language Model Agents","zh_title":"基于物理信息的大语言模型智能体用于化工过程PID整定框架","primary_category":"eess.SY","date":"2026-07-30","score":0,"bucket":"other","tags":["PID整定","大语言模型","过程控制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26594","has_summary":false},{"id":"2607.26109","title":"The Attention-Directing Ability of Teams","zh_title":"团队的注意力引导能力","primary_category":"physics.soc-ph","date":"2026-07-30","score":0,"bucket":"other","tags":["团队协调","注意力动态","人类行为"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26109","has_summary":false},{"id":"2607.26387","title":"\"Nobody Did This\": Contribution, Originality, and Accountability in Agent-Mediated Collaboration","zh_title":"“没人做过这个”：智能体中介协作中的贡献、原创性与问责","primary_category":"cs.CY","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体协作","知识工作","问责机制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26387","has_summary":false},{"id":"2607.26599","title":"Uncertainty-Guided LLM Semantic Augmentation for Heterogeneous Treatment Effect Estimation","zh_title":"不确定性引导的LLM语义增强用于异质性处理效应估计","primary_category":"cs.LG","date":"2026-07-30","score":0,"bucket":"other","tags":["因果推断","表示学习","异质性处理效应"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.26599","has_summary":false},{"id":"2607.26922","title":"Two Calls Beat Five Agents: Evaluating Multi-Agent Pipelines Against Self-Refinement for Local Language Models","zh_title":"两次调用胜过五个智能体：评估本地语言模型的多智能体流水线与自我优化","primary_category":"cs.LG","date":"2026-07-30","score":0,"bucket":"other","tags":["多智能体系统","推理优化","本地模型部署"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26922","has_summary":false},{"id":"2607.26091","title":"The Evolutionary Dynamics of AI, Politicization, Contestation, and Trust in Science Funding","zh_title":"人工智能、政治化、争议与信任在科学资助中的演化动力学","primary_category":"physics.soc-ph","date":"2026-07-30","score":0,"bucket":"other","tags":["演化博弈","科学政策","多智能体模拟"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26091","has_summary":false},{"id":"2607.25292","title":"Instruction-Tuned Language Models Cannot Sample from Distributions They Can Describe","zh_title":"指令微调语言模型无法从它们能描述的分布中采样","primary_category":"cs.AI","date":"2026-07-29","score":10,"bucket":"selected","tags":["LLM人类仿真","分布采样失效","算法保真度"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.25292","has_summary":true},{"id":"2607.24782","title":"Personalization, Personas, and Forecasting in Value Alignment","zh_title":"价值对齐中的个性化、角色与预测","primary_category":"cs.AI","date":"2026-07-29","score":9,"bucket":"selected","tags":["LLM仿真","价值观调查","文化对齐"],"rubric_hits":["A1","A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2607.24782","has_summary":true},{"id":"2607.25447","title":"CoRenew: A large language model agent-based policy simulation platform for multifamily residential redevelopment","zh_title":"CoRenew：基于大语言模型代理的多户住宅再开发政策仿真平台","primary_category":"cs.MA","date":"2026-07-29","score":9,"bucket":"selected","tags":["LLM仿真","政策评估","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.25447","has_summary":true},{"id":"2607.24765","title":"Measuring and Improving Behavioral Consistency in Large Language Models through Fact-Heuristic-Emotion State Enforcement","zh_title":"通过事实-启发-情感状态强制测量与提升大语言模型行为一致性","primary_category":"cs.CL","date":"2026-07-29","score":5,"bucket":"other","tags":["行为一致性","提示工程","模型评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.24765","has_summary":true},{"id":"2607.24999","title":"CogArena: A Multimethod Evaluation of Cognitive Ability Structure in Large Language Models","zh_title":"CogArena：大语言模型认知能力结构的多方法评估","primary_category":"cs.CL","date":"2026-07-29","score":5,"bucket":"other","tags":["认知评估","LLM能力结构","基准测试"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.24999","has_summary":true},{"id":"2607.26015","title":"Instruction-Tuned Models Locally Reuse Human Syntax More Than Humans Do","zh_title":"指令微调模型比人类更局部地复用人类句法","primary_category":"cs.CL","date":"2026-07-29","score":5,"bucket":"other","tags":["句法趋同","语言对齐","模型行为分析"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.26015","has_summary":true},{"id":"2607.25140","title":"How Affect Propagates among LLM Agents: Emergent Emotional Contagion in Crowd Simulation","zh_title":"情感如何在LLM智能体间传播：群体模拟中的涌现情绪传染","primary_category":"cs.AI","date":"2026-07-29","score":5,"bucket":"other","tags":["LLM智能体","情绪传染","群体模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.25140","has_summary":true},{"id":"2607.25485","title":"PatientAgentBench: A Benchmark Framework for Evaluating Patient-Facing Health AI Agents","zh_title":"PatientAgentBench：面向患者健康AI智能体的基准评估框架","primary_category":"cs.AI","date":"2026-07-29","score":5,"bucket":"other","tags":["医疗AI评估","LLM模拟患者","基准测试"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.25485","has_summary":true},{"id":"2607.25726","title":"Nudging Sustainable Choices through LLM-Generated Recommendation Explanations","zh_title":"通过LLM生成的推荐解释助推可持续选择","primary_category":"cs.AI","date":"2026-07-29","score":5,"bucket":"other","tags":["推荐系统","行为助推","可持续消费"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.25726","has_summary":true},{"id":"2607.25526","title":"Estimating the Geopolitical Preferences of Large Language Models from United Nations Voting Data","zh_title":"从联合国投票数据估计大语言模型的地缘政治偏好","primary_category":"cs.CY","date":"2026-07-29","score":5,"bucket":"other","tags":["LLM立场测量","联合国投票","地缘政治偏好"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.25526","has_summary":true},{"id":"2607.25019","title":"Interactive Alignment","zh_title":"交互式对齐","primary_category":"econ.TH","date":"2026-07-29","score":5,"bucket":"other","tags":["LLM仿真","演化博弈","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.25019","has_summary":true},{"id":"2607.25218","title":"Everyone is unique: Towards Behaviorally Heterogeneous Negotiation Dialogue Systems for Debt Collection","zh_title":"人人皆独特：面向催收的行为异质谈判对话系统","primary_category":"cs.AI","date":"2026-07-29","score":3,"bucket":"other","tags":["谈判对话系统","行为异质性","催收"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.25218","has_summary":false},{"id":"2607.21534","title":"Generative AI Availability, Grades, and Student Satisfaction at a Large University","zh_title":"生成式AI可用性、成绩与学生满意度：基于一所大型大学的研究","primary_category":"cs.CY","date":"2026-07-29","score":0,"bucket":"other","tags":["生成式AI","高等教育","成绩分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21534","has_summary":false},{"id":"2607.21859","title":"EviDAG: Auditable Causal DAG Authoring with Biomedical Literature","zh_title":"DAGForge：基于生物医学文献的可审计因果DAG构建","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["因果图构建","文献推理","生物医学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21859","has_summary":false},{"id":"2607.24750","title":"TimeCapsule: Generative Hallucination as a Method for Historical Sensemaking","zh_title":"时间胶囊：生成式幻觉作为历史意义建构的方法","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["历史文本生成","语言模型","诠释学"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.24750","has_summary":false},{"id":"2607.25184","title":"A scaling law of contextual persistence in human language","zh_title":"人类语言中上下文持久性的标度律","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["语言统计规律","标度律","LLM作为探针"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.25184","has_summary":false},{"id":"2607.25202","title":"A Cross-lingual Comparison of Human and Classification Model Entrainment Behavior in Code-switched Speech Settings","zh_title":"语码转换语音中人类与分类模型entrainment行为的跨语言比较","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["对话行为分析","语码转换","分类模型"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.25202","has_summary":false},{"id":"2607.25308","title":"CAST: Game Solvers as Turn-Level Teachers for LLM Agents","zh_title":"CAST：将游戏求解器作为回合级教师用于LLM智能体","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["多智能体","游戏求解","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25308","has_summary":false},{"id":"2607.25375","title":"Inspect India Evals: An Open Benchmarking Framework for Evaluating Large Language Models in the Indian Linguistic and Cultural Context","zh_title":"Inspect India Evals：评估印度语言文化背景下大语言模型的开放基准框架","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["LLM评测","多语言基准","文化偏见"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.25375","has_summary":false},{"id":"2607.25881","title":"AI's Capability in Assisting Scientific Research in Physics, Astrophysics, and Cosmology II: Project Planning and Proposal Evaluation","zh_title":"AI在物理、天体物理和宇宙学中辅助科学研究的能力II：项目规划与提案评估","primary_category":"cs.CL","date":"2026-07-29","score":0,"bucket":"other","tags":["LLM能力评测","科学项目规划","提案评审"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.25881","has_summary":false},{"id":"2607.24768","title":"PATHFinder Agent for Tailored Prenatal Care","zh_title":"用于定制化产前护理的PATHFinder智能体","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["对话系统","临床决策支持","产前护理"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.24768","has_summary":false},{"id":"2607.24769","title":"LLM Scheming Inversely Scales with Pretraining Language Coverage","zh_title":"LLM的欺骗行为与预训练语言覆盖率呈反向缩放","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["AI安全","多语言评测","模型行为分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24769","has_summary":false},{"id":"2607.24817","title":"Retrieval-Augmented Generation in LLMs for Mental Health: Quantifying the Incremental Contribution of Retrieval Within a Layered Safety Architecture","zh_title":"大语言模型中的检索增强生成用于心理健康：量化分层安全架构中检索的增量贡献","primary_category":"cs.IR","date":"2026-07-29","score":0,"bucket":"other","tags":["心理健康聊天机器人","检索增强生成","意图检测"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.24817","has_summary":false},{"id":"2607.25340","title":"Cardiologent: Multi-Agent Clinical Decision Support for Patient-Level Arrhythmia Assessment, Urgency, and Management","zh_title":"Cardiologent：面向患者级心律失常评估、紧急程度与管理的多智能体临床决策支持","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["多智能体系统","临床决策支持","心律失常"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25340","has_summary":false},{"id":"2607.25057","title":"Psychological Influences of Conversational AI: Research and Design Directions for Reducing Harm and Promoting Well-Being","zh_title":"对话式AI的心理影响：减少伤害与促进福祉的研究与设计方向","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["对话AI","用户福祉","心理影响"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.25057","has_summary":false},{"id":"2607.25152","title":"When Do Agent Loops Mistake Stagnation for Progress? Self-Evaluation Bias and Externally Grounded Verification in Long-Running Autonomous LLM Agent Loops","zh_title":"代理循环何时将停滞误认为进展？长期自主LLM代理循环中的自我评估偏差与外部验证","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["LLM代理","自我评估偏差","自主循环"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25152","has_summary":false},{"id":"2607.25279","title":"Many-body Tipping Dynamics of ChatGPT-like AIs","zh_title":"类ChatGPT人工智能的多体倾覆动力学","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["AI故障分析","多体动力学","token交互"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25279","has_summary":false},{"id":"2607.25446","title":"Toward an Organizational Science of Multi-Agent LLM Systems: Decoupling Who, How, and Which Algorithm","zh_title":"迈向多智能体LLM系统的组织科学：解耦谁、如何和哪种算法","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["多智能体系统","组织设计","协作协议"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25446","has_summary":false},{"id":"2607.25620","title":"Beyond Epistemia: Epistemic Schizologia and Large Language Models as Techno-Semiotic Machines","zh_title":"超越Epistemia：认识论分裂症与作为技术符号机器的大语言模型","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["认识论","符号学","人机交互哲学"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.25620","has_summary":false},{"id":"2607.25877","title":"Runtime Uncertainty Monitoring for LLM-Based Multi-Agent Systems Using Bayesian Networks","zh_title":"基于贝叶斯网络的LLM多智能体系统运行时不确定性监控","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["多智能体系统","不确定性量化","精算建模"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25877","has_summary":false},{"id":"2607.26034","title":"Falling Behind Drives Unsafe Development in an Idealised AI Race Experiment","zh_title":"落后驱动理想化AI竞赛实验中的不安全开发","primary_category":"cs.AI","date":"2026-07-29","score":0,"bucket":"other","tags":["行为实验","AI竞赛","风险决策"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.26034","has_summary":false},{"id":"2607.24749","title":"Game AI Not Fun? A Scoping Review and Meta-Analysis on the Differences in Enjoyment between Human and Computer Opponents","zh_title":"游戏AI不好玩？人类与电脑对手乐趣差异的范围综述与元分析","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["游戏AI","玩家体验","元分析"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.24749","has_summary":false},{"id":"2607.24761","title":"Verification Without Distrust: Reframing User-Side Oversight as Routine Epistemic Governance in Everyday Human-Chatbot Interaction","zh_title":"无需不信任的验证：将用户端监督重新定义为日常人机对话交互中的常规认知治理","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["人机交互","用户行为","信任与验证"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.24761","has_summary":false},{"id":"2607.25257","title":"Laplace-PSN-IRT: Uncertainty Quantification for Neural Item Response Theory Models of LLM Benchmarks","zh_title":"Laplace-PSN-IRT：LLM 基准测试的神经项目反应理论模型的不确定性量化","primary_category":"stat.AP","date":"2026-07-29","score":0,"bucket":"other","tags":["项目反应理论","LLM 评测","不确定性量化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.25257","has_summary":false},{"id":"2607.24775","title":"Empathy and the Human-Moment Gaps of AI Chatbots: Insights from Empathy Displacement Theory","zh_title":"AI聊天机器人的同理心与人机时刻差距：基于同理心位移理论的见解","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["AI同理心","人机交互","概念框架"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.24775","has_summary":false},{"id":"2607.25131","title":"Beyond the Post Hoc User Study: Modeling Visual Decision-Making with Active Inference","zh_title":"超越事后用户研究：用主动推理建模视觉决策","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["认知建模","可视化评估","主动推理"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.25131","has_summary":false},{"id":"2607.25574","title":"\"Dragon Slayer Becomes the Dragon\": How Players Perceive and Respond to Inequality in the Game World of Whiteout Survival","zh_title":"“屠龙者终成恶龙”：玩家如何感知和应对《Whiteout Survival》游戏世界中的不平等","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["游戏研究","不平等感知","玩家行为"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.25574","has_summary":false},{"id":"2607.25922","title":"Faster, Higher, Stronger? The Impact of GenAI on Knowledge Work Productivity - Evidence from the Field","zh_title":"更快、更高、更强？生成式AI对知识工作生产力的影响——来自现场的实证","primary_category":"cs.HC","date":"2026-07-29","score":0,"bucket":"other","tags":["生成式AI","生产力","现场实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25922","has_summary":false},{"id":"2607.25240","title":"From Compressing Complexity to Accommodating Complexity: How AI Transforms Standardization and Individualization","zh_title":"从压缩复杂性到容纳复杂性：AI如何改变标准化与个性化","primary_category":"cs.CY","date":"2026-07-29","score":0,"bucket":"other","tags":["AI与社会","标准化","信息处理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25240","has_summary":false},{"id":"2607.25514","title":"Learning Dynamics of Strategic Publishers in Generative AI Ecosystems","zh_title":"生成式AI生态系统中战略出版商的学习动态","primary_category":"cs.GT","date":"2026-07-29","score":0,"bucket":"other","tags":["博弈论","多智能体系统","生成式AI搜索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25514","has_summary":false},{"id":"2607.25472","title":"Algorithm-Driven Information Similarity and Collective Action: An Experimental Study","zh_title":"算法驱动的信息相似性与集体行动：一项实验研究","primary_category":"econ.GN","date":"2026-07-29","score":0,"bucket":"other","tags":["集体行动","信息相似性","人类实验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.25472","has_summary":false},{"id":"2604.02458","title":"Statistical realism is not evidence that LLMs can estimate treatment effects in social science experiments","zh_title":"统计真实性不能证明LLM能估计社会科学实验中的处理效应","primary_category":"cs.CY","date":"2026-07-28","score":10,"bucket":"selected","tags":["LLM仿真","处理效应估计","统计真实性"],"rubric_hits":["A1","A2","A4","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.02458","has_summary":true},{"id":"2607.22605","title":"Socioeconomic Inference in LLM Medical Triage: Same Symptoms, Different ZIP Code","zh_title":"大语言模型医疗分诊中的社会经济推断：相同症状，不同邮编","primary_category":"cs.CY","date":"2026-07-28","score":9,"bucket":"selected","tags":["LLM仿真","医疗决策偏差","社会经济地位"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.22605","has_summary":true},{"id":"2607.23037","title":"Speech Signals Complement LLMs for Predicting Interpersonal Attraction in Speed Dating","zh_title":"语音信号补充大语言模型预测速配中的人际吸引","primary_category":"cs.CL","date":"2026-07-28","score":5,"bucket":"other","tags":["人际吸引预测","多模态融合","LLM标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2607.23037","has_summary":true},{"id":"2607.23976","title":"Tag Questions and the Generational Reversal of Sycophancy Across 45 Language Models","zh_title":"附加疑问句与45个语言模型逢迎倾向的代际逆转","primary_category":"cs.CL","date":"2026-07-28","score":5,"bucket":"other","tags":["LLM行为测量","逢迎倾向","代际变化"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.23976","has_summary":true},{"id":"2607.23519","title":"Auditing Alignment Controllability in LLMs via Political Axes","zh_title":"通过政治轴审计大语言模型的对齐可控性","primary_category":"cs.CY","date":"2026-07-28","score":5,"bucket":"other","tags":["LLM审计","政治立场","可控性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.23519","has_summary":true},{"id":"2607.22513","title":"Opaque Epistemic Mediation: How LLM Deployment Configurations Shape the Validation of Pseudo-Science","zh_title":"不透明的认知中介：LLM部署配置如何塑造伪科学的验证","primary_category":"cs.CY","date":"2026-07-28","score":5,"bucket":"other","tags":["LLM立场测量","伪科学验证","部署配置影响"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.22513","has_summary":true},{"id":"2607.23993","title":"On Capturing the Narrative: Social Media Manipulation Wargaming for Cyberliteracy","zh_title":"捕捉叙事：面向网络素养的社交媒体操纵兵棋推演","primary_category":"cs.CY","date":"2026-07-28","score":5,"bucket":"other","tags":["社会模拟","LLM智能体","虚假信息"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2607.23993","has_summary":true},{"id":"2607.21596","title":"FlowEvo: Self-Evolving Agents through the Co-Evolution of Workflows and Executable Skills","zh_title":"FlowEvo：通过工作流与可执行技能的协同进化实现自我进化的智能体","primary_category":"cs.AI","date":"2026-07-28","score":2,"bucket":"other","tags":["多智能体","工作流进化","技能库"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21596","has_summary":false},{"id":"2607.21616","title":"Lost in Context: Addressing Context Anxiety in Large Language Models","zh_title":"迷失在上下文中：解决大语言模型中的上下文焦虑","primary_category":"cs.AI","date":"2026-07-28","score":2,"bucket":"other","tags":["LLM推理","上下文焦虑","模型能力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21616","has_summary":false},{"id":"2607.22014","title":"Zero-Shot Mission-Level Evaluation for Aerial MLLM Agents","zh_title":"空中多模态大语言模型智能体的零样本任务级评估","primary_category":"cs.AI","date":"2026-07-28","score":2,"bucket":"other","tags":["空中机器人","多模态大模型","任务评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.22014","has_summary":false},{"id":"2607.22083","title":"Nanbeige4.2-3B: Unlocking Agentic Capabilities in a Compact Model","zh_title":"Nanbeige4.2-3B：解锁紧凑模型中的智能体能力","primary_category":"cs.AI","date":"2026-07-28","score":2,"bucket":"other","tags":["智能体模型","工具使用","代码智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22083","has_summary":false},{"id":"2607.21612","title":"Procedural Knowledge Is Not Low-Rank: Why LoRA Fails to Internalize Multi-Step Procedures","zh_title":"程序性知识不是低秩的：为什么LoRA无法内化多步骤程序","primary_category":"cs.AI","date":"2026-07-28","score":1,"bucket":"other","tags":["参数高效微调","LoRA","程序性知识"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21612","has_summary":false},{"id":"2607.21613","title":"The Hard Decision Layer: Evidence for Committed Inference in Transformers","zh_title":"硬决策层：Transformer中承诺推理的证据","primary_category":"cs.AI","date":"2026-07-28","score":1,"bucket":"other","tags":["模型内部机制","多项选择问答","推理效率"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21613","has_summary":false},{"id":"2607.22553","title":"Evaluating the Impact of Reviewer Guideline Design on LLM-Based Automated Peer Review","zh_title":"评估审稿指南设计对基于LLM的自动同行评审的影响","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["自动同行评审","LLM评测","审稿指南"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22553","has_summary":false},{"id":"2607.23083","title":"LoRA for Gender-Inclusive Rewriting and Activation Steering for Counter-Narrative Generation","zh_title":"用于性别包容重写的LoRA与用于反叙事生成的激活引导","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["包容性语言生成","激活引导","反叙事生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23083","has_summary":false},{"id":"2607.23440","title":"Reasoning or Memorization: Can LLMs Understand and Generate Chinese Xiehouyu Riddles?","zh_title":"推理还是记忆：大语言模型能理解和生成中文歇后语谜题吗？","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["NLP评测","语言游戏","数据污染"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23440","has_summary":false},{"id":"2607.23513","title":"Do Diagrams Help Large Language Models Reason? Evidence from Syllogistic Reasoning","zh_title":"图表能帮助大语言模型推理吗？来自三段论推理的证据","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM推理","图表辅助","逻辑推理"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23513","has_summary":false},{"id":"2607.23538","title":"Guiding Language Models to Be More Empathetic: Culturally Sensitive Mental Health Advice Generation Through Human-LLM Collaboration","zh_title":"引导语言模型更具共情力：通过人机协作生成文化敏感的心理健康建议","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["心理健康","角色扮演","提示工程"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.23538","has_summary":false},{"id":"2607.23915","title":"Understanding Tone-Dependent Inference Cost in Large Language Models","zh_title":"理解大语言模型中语气依赖的推理成本","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["提示工程","推理成本","MMLU评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23915","has_summary":false},{"id":"2607.24072","title":"LLM-Based vs. Lexicon-Based Sentiment Signals for Tail-Risk Detection in Meme Stocks","zh_title":"基于LLM与基于词典的情感信号在模因股尾部风险检测中的比较","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["情感分析","金融NLP","模因股"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24072","has_summary":false},{"id":"2607.24300","title":"Self-Authored Verification Is Unreliable in Heuristic Self-Improving Agents","zh_title":"启发式自我改进智能体中自编验证不可靠","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","自我改进","验证可靠性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24300","has_summary":false},{"id":"2607.24352","title":"Retrieval-Augmented Large Language Models as Components of Cognitive Computing architecture for Regulatory Knowledge Management","zh_title":"检索增强大语言模型作为认知计算架构组件用于法规知识管理","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["RAG","认知计算","法规管理"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24352","has_summary":false},{"id":"2607.24368","title":"Keep It InMind: Benchmarking the Implicit-Association Blind Spot in Agent Memory","zh_title":"铭记于心：评估智能体记忆中内隐联想盲区的基准测试","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["记忆系统","基准测试","知识检索"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24368","has_summary":false},{"id":"2607.24471","title":"Grounding latent algorithm routing in transformer reasoning","zh_title":"在Transformer推理中扎根潜在算法路由","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["Transformer","上下文学习","算法路由"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24471","has_summary":false},{"id":"2607.22554","title":"Same Question, Different Answers: Evaluating LLM Reliability Beyond Accuracy","zh_title":"相同问题，不同答案：超越准确率评估大语言模型的可靠性","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM可靠性","一致性评估","提示敏感性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22554","has_summary":false},{"id":"2607.22676","title":"How LLM Task-Adaptation Reshapes Alignment: A Multi-dimensional Study of Behavioral and Representational Drift","zh_title":"LLM任务适应如何重塑对齐：行为与表征漂移的多维研究","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["对齐评估","任务适应","表征分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22676","has_summary":false},{"id":"2607.23927","title":"Reality Monitoring in Large Language Models: Self-Knowledge That Transforms with Conversation Memory","zh_title":"大语言模型中的现实监控：随对话记忆转变的自我认知","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["源记忆","认知评测","幻觉"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23927","has_summary":false},{"id":"2607.24339","title":"Gubernaut: A Deterministic Homeostatic Controller for Affect-Regulated LLM Agents, Validated Across Independent Model Families","zh_title":"Gubernaut：一种用于情感调节LLM智能体的确定性稳态控制器，跨独立模型家族验证","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM智能体","情感调节","运行时控制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24339","has_summary":false},{"id":"2607.24484","title":"What do Reward Models Memorize?","zh_title":"奖励模型记住了什么？","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["奖励模型","记忆分析","偏好数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24484","has_summary":false},{"id":"2607.21606","title":"TILT: Improving Compositional Generation in Diffusion Models with a Model-Intrinsic Reward","zh_title":"TILT：利用模型内在奖励改进扩散模型中的组合生成","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["图像生成","扩散模型","组合生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21606","has_summary":false},{"id":"2607.21933","title":"Semiotic logical hexagon theory for LLM logical reasoning","zh_title":"用于大语言模型逻辑推理的符号逻辑六边形理论","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["逻辑推理","语义组织","LLM评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21933","has_summary":false},{"id":"2607.22520","title":"The Regression Tax: Decomposing Why Skills Help and Hurt LLM Agents","zh_title":"回归税：分解技能为何帮助和损害LLM智能体","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","任务成功率","技能评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22520","has_summary":false},{"id":"2607.21186","title":"Do emulated quantum circuits change what CNNs look at? Performance and explainability comparison in medical image classification","zh_title":"模拟量子电路会改变CNN的观察方式吗？医学图像分类中的性能与可解释性比较","primary_category":"quant-ph","date":"2026-07-28","score":0,"bucket":"other","tags":["量子机器学习","医学图像分类","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21186","has_summary":false},{"id":"2607.21598","title":"Control panels to clarify user intent with Large Language Models","zh_title":"用控制面板澄清大语言模型的用户意图","primary_category":"cs.HC","date":"2026-07-28","score":0,"bucket":"other","tags":["人机交互","用户界面","提示工程"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.21598","has_summary":false},{"id":"2607.21599","title":"Decoupled Attention Fusion: Accelerating RAG with Efficient KV Cache Reuse","zh_title":"解耦注意力融合：通过高效KV缓存重用加速RAG","primary_category":"cs.PF","date":"2026-07-28","score":0,"bucket":"other","tags":["RAG","KV缓存","推理加速"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21599","has_summary":false},{"id":"2607.21603","title":"Analyzing Middle School Students' Dialogue and Behaviors during Collaborative AI Chatbot Development Using Ordered Network Analysis","zh_title":"使用有序网络分析分析中学生在协作AI聊天机器人开发中的对话和行为","primary_category":"cs.HC","date":"2026-07-28","score":0,"bucket":"other","tags":["AI教育","协作学习","有序网络分析"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.21603","has_summary":false},{"id":"2607.21608","title":"From Obligation to Specification: A Survey on Validating EU AI Act Requirements in RE","zh_title":"从义务到规范：验证欧盟AI法案需求工程要求的综述","primary_category":"cs.SE","date":"2026-07-28","score":0,"bucket":"other","tags":["需求工程","合规验证","LLM工具"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21608","has_summary":false},{"id":"2607.21656","title":"Cross-Model LLM Code Review: Should you use Claude to review Codex or vice versa?","zh_title":"跨模型大语言模型代码审查：应该用Claude审查Codex还是反之？","primary_category":"cs.SE","date":"2026-07-28","score":0,"bucket":"other","tags":["代码审查","多智能体协作","LLM评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21656","has_summary":false},{"id":"2607.21774","title":"Probing Latent Colombian Identity Inferences in Qwen2.5-7B with Natural Language Autoencoders","zh_title":"用自然语言自编码器探测Qwen2.5-7B中的潜在哥伦比亚身份推断","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["模型可解释性","偏见探测","表征分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21774","has_summary":false},{"id":"2607.21964","title":"ACME: A Multi-Cultural, Multi-Embodiment Social-Navigation Dataset","zh_title":"ACME：一个多文化、多形态的社交导航数据集","primary_category":"cs.RO","date":"2026-07-28","score":0,"bucket":"other","tags":["社交导航","机器人","数据集"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.21964","has_summary":false},{"id":"2607.21981","title":"J-CoT: Chain-of-Thought in J-Space","zh_title":"J-CoT：J空间中的思维链","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["思维链","潜在推理","语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.21981","has_summary":false},{"id":"2607.22100","title":"MEUSLI: a Multilingual Projector for LLM-based ASR and Beyond","zh_title":"MEUSLI：用于基于LLM的ASR及更多任务的多语言投影器","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["语音识别","多语言","投影器"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22100","has_summary":false},{"id":"2607.22182","title":"From Isolated Tasks to Structured Capabilities: A Multilayer Taxonomy for Large Language Models","zh_title":"从孤立任务到结构化能力：大语言模型的多层分类法","primary_category":"cs.CL","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM评估","能力分类","研究组织"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22182","has_summary":false},{"id":"2607.22428","title":"Unboxing Diffusion Models for the Arts: Interactive Model Bending and Practice-Based Explainability","zh_title":"为艺术揭开扩散模型的面纱：交互式模型弯曲与基于实践的的可解释性","primary_category":"cs.HC","date":"2026-07-28","score":0,"bucket":"other","tags":["可解释AI","扩散模型","创意实践"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.22428","has_summary":false},{"id":"2607.23489","title":"Multimodal Data Comprehension: Understanding How Visual-Textual Chains of Information Influence Data Interpretation","zh_title":"多模态数据理解：视觉-文本信息链如何影响数据解读","primary_category":"cs.HC","date":"2026-07-28","score":0,"bucket":"other","tags":["多模态理解","数据可视化","人类认知"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23489","has_summary":false},{"id":"2607.24360","title":"Modeling Duelling Contagions of True and False Information in the Face of Inherent Individual biases","zh_title":"面对固有个体偏见时真假信息竞争传播的建模","primary_category":"cs.SI","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体模型","信息扩散","认知偏差"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24360","has_summary":false},{"id":"2607.23336","title":"Constitutional governance for societies of AI agents in the built environment: a research agenda","zh_title":"建筑环境中AI智能体社会的宪政治理：研究议程","primary_category":"cs.CY","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","建筑环境","治理机制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23336","has_summary":false},{"id":"2607.23931","title":"State-dependent error correlations shape voting thresholds in committees of AI agents","zh_title":"状态依赖的误差相关性塑造AI代理委员会中的投票阈值","primary_category":"cs.CY","date":"2026-07-28","score":0,"bucket":"other","tags":["AI委员会","投票机制","误差相关性"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23931","has_summary":false},{"id":"2607.22758","title":"Spectral Dynamics of Semantic Drift in Clinical Multi-Agent Language Model Networks","zh_title":"临床多智能体语言模型网络中语义漂移的谱动力学","primary_category":"cs.MA","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","语义漂移","网络拓扑"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22758","has_summary":false},{"id":"2607.23311","title":"Emergent Behaviour in Financial Markets","zh_title":"金融市场中的涌现行为","primary_category":"cs.MA","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","涌现行为","金融市场"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23311","has_summary":false},{"id":"2607.24416","title":"Decentralised Consensus Learning Networks: SME Rotation Without Centralised Reward","zh_title":"去中心化共识学习网络：无中心化奖励的SME轮换","primary_category":"cs.MA","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","共识学习","去中心化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24416","has_summary":false},{"id":"2607.22591","title":"Lexical discovery in unknown environments orchestrated by Large Language Models","zh_title":"大语言模型驱动的未知环境词汇发现","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","词汇涌现","自主探索"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.22591","has_summary":false},{"id":"2607.23197","title":"Domain-Prior-Regularized Graph Modeling for Anomaly Detection in Cyber-Physical Systems","zh_title":"基于领域先验正则化图建模的信息物理系统异常检测","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["异常检测","信息物理系统","图神经网络"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.23197","has_summary":false},{"id":"2607.23333","title":"Training with (Swap) Regret Loss in a Single-Layer Self-Attention Model: A Case Study on the Probability Simplex","zh_title":"单层自注意力模型中的（交换）遗憾损失训练：概率单纯形案例研究","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["博弈论","在线学习","自注意力机制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23333","has_summary":false},{"id":"2607.23488","title":"Learning Sampling Parameters for Diffusion Models","zh_title":"学习扩散模型的采样参数","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["扩散模型","强化学习","文本到图像生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23488","has_summary":false},{"id":"2607.23647","title":"CALMRec: Causally Aligned Language Memory for Long-Horizon Recommendation","zh_title":"因果对齐语言记忆的长周期推荐框架","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["推荐系统","因果推断","语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23647","has_summary":false},{"id":"2607.24425","title":"Context Is King: How In-Context Specification Shapes the Geometry of Concepts","zh_title":"上下文为王：上下文规范如何塑造概念几何","primary_category":"cs.LG","date":"2026-07-28","score":0,"bucket":"other","tags":["概念几何","上下文学习","模型可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24425","has_summary":false},{"id":"2607.22561","title":"Codifying the Judge: Scalable Evaluation via Program Distillation","zh_title":"编码法官：通过程序蒸馏实现可扩展评估","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["自动评估","程序蒸馏","LLM-as-a-judge"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22561","has_summary":false},{"id":"2607.22646","title":"Extracting Algorithms in Pre-trained LLMs: A Case on Hidden Markov Models","zh_title":"从预训练大语言模型中提取算法：以隐马尔可夫模型为例","primary_category":"cs.AI","date":"2026-07-28","score":0,"bucket":"other","tags":["机制可解释性","上下文学习","隐马尔可夫模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22646","has_summary":false},{"id":"2607.22880","title":"Do Coverage and Mutation Scores of LLM-Generated Test Suites Correlate with Their Effectiveness? (Replicability Study)","zh_title":"LLM生成测试套件的覆盖率和变异分数与其有效性相关吗？（可复现性研究）","primary_category":"cs.SE","date":"2026-07-28","score":0,"bucket":"other","tags":["软件测试","LLM生成测试","覆盖率与变异分数"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22880","has_summary":false},{"id":"2607.22951","title":"Modeling Memory-Dependent Reliability of LLMs: A Hidden Markov Model","zh_title":"建模大语言模型的记忆依赖可靠性：一种隐马尔可夫模型","primary_category":"stat.ML","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM可靠性","统计推断","序列依赖"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22951","has_summary":false},{"id":"2607.23075","title":"Traceable LLM Reasoning for Fake-Order Fraud Detection","zh_title":"面向虚假订单欺诈检测的可追溯大语言模型推理","primary_category":"cs.CR","date":"2026-07-28","score":0,"bucket":"other","tags":["欺诈检测","强化学习","大语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23075","has_summary":false},{"id":"2607.23983","title":"HydroAgent: Formalizing Forecaster Expertise into Skill-Orchestrated Flood Forecasting Workflows","zh_title":"HydroAgent：将预报员专业知识形式化为技能编排的洪水预报工作流","primary_category":"physics.geo-ph","date":"2026-07-28","score":0,"bucket":"other","tags":["洪水预报","LLM智能体","工作流编排"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23983","has_summary":false},{"id":"2607.24392","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","zh_title":"当LLM防御适得其反：安全、性能与成本的权衡特征分析","primary_category":"cs.CR","date":"2026-07-28","score":0,"bucket":"other","tags":["LLM安全","防御机制","性能评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.24392","has_summary":false},{"id":"2607.24401","title":"proxymate: Diagnosis and Adjustment of Proxy Estimates for Reliable Inference","zh_title":"proxymate：代理估计的诊断与调整以实现可靠推断","primary_category":"stat.ML","date":"2026-07-28","score":0,"bucket":"other","tags":["代理指标","统计推断","数据验证"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24401","has_summary":false},{"id":"2607.23739","title":"Separating Clicks from Baits: Using Large Language Models to Detect Misleading YouTube Thumbnails","zh_title":"区分点击与诱饵：使用大语言模型检测误导性YouTube缩略图","primary_category":"cs.SI","date":"2026-07-28","score":0,"bucket":"other","tags":["内容审核","多模态检测","LLM应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.23739","has_summary":false},{"id":"2607.24698","title":"Modest Algorithmic Mediation can Maximize Topical Diversity in Hybrid Human-AI Systems","zh_title":"适度算法中介可最大化混合人机系统中的主题多样性","primary_category":"cs.SI","date":"2026-07-28","score":0,"bucket":"other","tags":["信息扩散","算法推荐","社交网络"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24698","has_summary":false},{"id":"2607.23168","title":"Individual and collective gains from cooperation and reciprocity in a dynamic-network Prisoner's Dilemma driven by extraversion, openness, and agreeableness","zh_title":"外向性、开放性和宜人性驱动的动态网络囚徒困境中合作与互惠的个体与集体收益","primary_category":"physics.soc-ph","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体模拟","囚徒困境","人格特质"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23168","has_summary":false},{"id":"2607.24175","title":"A World of Ginis","zh_title":"基尼系数的世界","primary_category":"econ.GN","date":"2026-07-28","score":0,"bucket":"other","tags":["经济不平等","基尼系数","数据协调"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2607.24175","has_summary":false},{"id":"2607.24389","title":"How to Disrupt a Market","zh_title":"如何扰乱一个市场","primary_category":"econ.GN","date":"2026-07-28","score":0,"bucket":"other","tags":["市场设计","网络实验","非法市场"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.24389","has_summary":false},{"id":"2607.23254","title":"Towards Optimal Estimators for Randomized Control Trials","zh_title":"迈向随机对照试验的最优估计量","primary_category":"stat.AP","date":"2026-07-28","score":0,"bucket":"other","tags":["因果推断","随机对照试验","估计量选择"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23254","has_summary":false},{"id":"2607.24143","title":"Inference on counterfactual distributions using martingale posteriors","zh_title":"使用鞅后验进行反事实分布的推断","primary_category":"stat.ME","date":"2026-07-28","score":0,"bucket":"other","tags":["因果推断","反事实分布","鞅后验"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24143","has_summary":false},{"id":"2607.24304","title":"Teacher Knows It Best: Spontaneous Symmetry Breaking and Tipping Points in Networked Langevin Dynamics AI Sycophancy","zh_title":"老师最懂：网络化朗之万动力学AI谄媚中的自发对称破缺与临界点","primary_category":"physics.soc-ph","date":"2026-07-28","score":0,"bucket":"other","tags":["多智能体系统","统计物理","AI谄媚"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.24304","has_summary":false},{"id":"2607.21757","title":"Co-design of LLM-based preference agents: participation may drive overtrust","zh_title":"基于大语言模型的偏好代理协同设计：参与可能驱动过度信任","primary_category":"cs.CY","date":"2026-07-27","score":9,"bucket":"selected","tags":["LLM人类仿真","偏好代理","人机对齐"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.21757","has_summary":true},{"id":"2607.22218","title":"Why Large Language Models and Humans Converge and Diverge in Evaluating Creativity","zh_title":"大型语言模型与人类在创造力评估中为何趋同与分歧","primary_category":"cs.CL","date":"2026-07-27","score":7,"bucket":"pending","tags":["LLM评估","人机对齐","创造力判断"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.22218","has_summary":true},{"id":"2607.24372","title":"Randomness in large language models: What researchers need to know (and report)","zh_title":"大语言模型中的随机性：研究者须知（及应报告事项）","primary_category":"econ.GN","date":"2026-07-27","score":7,"bucket":"pending","tags":["LLM随机性","可重复性","报告标准"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2607.24372","has_summary":true},{"id":"2607.21614","title":"Household Movement Detection in Mixed-Format Occupancy Data Using LLM-Based Entity Resolution","zh_title":"基于LLM实体解析的混合格式居住数据中家庭迁移检测","primary_category":"cs.AI","date":"2026-07-27","score":0,"bucket":"other","tags":["实体解析","图推理","数据质量"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21614","has_summary":false},{"id":"2607.22305","title":"A Roadmap to Impactful Pluralistic Alignment Research","zh_title":"迈向有影响力的多元对齐研究路线图","primary_category":"cs.AI","date":"2026-07-27","score":0,"bucket":"other","tags":["多元对齐","AI治理","价值对齐"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.22305","has_summary":false},{"id":"2607.22345","title":"Teachy Mini: Development and Preliminary Evaluation of a Knowledge-Based Generative Social Robot for Higher Education","zh_title":"Teachy Mini：基于知识的高等教育生成式社交机器人开发与初步评估","primary_category":"cs.RO","date":"2026-07-27","score":0,"bucket":"other","tags":["社交机器人","教育辅导","人机交互"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2607.22345","has_summary":false},{"id":"2607.22463","title":"Beyond Perspectives: A Trio-Ethnography of Interpretation Evolution in LLM-Supported Programming Education","zh_title":"超越视角：LLM支持的编程教育中解释演变的三方民族志","primary_category":"cs.HC","date":"2026-07-27","score":0,"bucket":"other","tags":["编程教育","生成式AI","民族志"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2607.22463","has_summary":false},{"id":"2607.23313","title":"Agentic AI Orchestration of Heterogeneous Economic Models for Rapid, Multi-scenario Analysis of Energy Crises","zh_title":"基于异构经济模型的智能体AI编排用于能源危机快速多场景分析","primary_category":"econ.GN","date":"2026-07-25","score":0,"bucket":"other","tags":["多智能体系统","经济模型集成","能源危机分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.23313","has_summary":false},{"id":"2607.21268","title":"pAI-Econ-claude: A Gated Human-in-the-Loop Multi-Agent Architecture for AI-Assisted Economic Theory Development","zh_title":"pAI-Econ-claude：一种用于AI辅助经济学理论开发的门控人机协同多智能体架构","primary_category":"cs.MA","date":"2026-07-23","score":0,"bucket":"other","tags":["多智能体系统","经济学理论","人机协同"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.21268","has_summary":false},{"id":"2607.20410","title":"LKValues: Aligning Large Language Models with Sri Lankan Societal Values","zh_title":"LKValues：将大语言模型与斯里兰卡社会价值观对齐","primary_category":"cs.CL","date":"2026-07-22","score":5,"bucket":"other","tags":["价值观对齐","文化偏见","低资源语言"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.20410","has_summary":true},{"id":"2607.17437","title":"Empirical Grounding Improves the Realism of LLM Agents Simulating Human Behavior During Disruptions","zh_title":"经验锚定提升大语言模型代理在中断期间模拟人类行为的真实性","primary_category":"cs.AI","date":"2026-07-19","score":10,"bucket":"selected","tags":["人类行为仿真","经验锚定","灾害响应"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2607.17437","has_summary":true},{"id":"2607.18310","title":"Distribution-First Population Simulation: Collapse, Calibration, and Recall in Non-WEIRD LLM Persona Modeling","zh_title":"分布优先的人口模拟：非WEIRD LLM角色建模中的崩溃、校准与回忆","primary_category":"physics.soc-ph","date":"2026-07-17","score":10,"bucket":"selected","tags":["LLM人类仿真","分布校准","合成人口"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2607.18310","has_summary":true},{"id":"2607.14485","title":"Step-Level Preference Learning for Generative Agents in Social Simulations","zh_title":"面向社会模拟中生成式智能体的步级偏好学习","primary_category":"cs.AI","date":"2026-07-16","score":7,"bucket":"pending","tags":["LLM仿真","人类偏好学习","社会模拟"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2607.14485","has_summary":true},{"id":"2607.14371","title":"Supervised Fine-Tuning vs. In-Context Learning: An Equilibrium Analysis of LLM Personalization under Congestion","zh_title":"监督微调与上下文学习：拥塞下LLM个性化的均衡分析","primary_category":"cs.LG","date":"2026-07-15","score":0,"bucket":"other","tags":["LLM个性化","均衡分析","平台策略"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.14371","has_summary":false},{"id":"2607.12219","title":"Partial Identification with Multiple Nonlinear Measurements of a Latent Regressor","zh_title":"具有潜变量多个非线性测量的部分识别","primary_category":"econ.EM","date":"2026-07-13","score":0,"bucket":"other","tags":["计量经济学","部分识别","潜变量"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2607.12219","has_summary":false},{"id":"2607.11722","title":"STEP: Career-Path Recommendation via Temporal and Educational Trajectory Modeling","zh_title":"STEP：基于时序与教育轨迹建模的职业路径推荐","primary_category":"cs.CL","date":"2026-07-13","score":0,"bucket":"other","tags":["职业路径推荐","序列预测","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.11722","has_summary":false},{"id":"2607.08681","title":"SolarChain-Eval: A Physics-Constrained Benchmark for Trustworthy Economic Agents in Decentralized Energy Markets","zh_title":"SolarChain-Eval：面向去中心化能源市场中可信经济智能体的物理约束基准","primary_category":"cs.AI","date":"2026-07-09","score":0,"bucket":"other","tags":["多智能体系统","能源市场","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2607.08681","has_summary":false},{"id":"2607.06080","title":"From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations","zh_title":"从蓝图到现实：基于LLM多智能体仿真建模与应用普特南社会资本理论","primary_category":"cs.CL","date":"2026-07-07","score":9,"bucket":"selected","tags":["LLM人类仿真","社会资本理论","集体行动实验"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.06080","has_summary":true},{"id":"2607.05861","title":"Mitigating Factual Hallucination in Large Reasoning Models via Mixed-Mode Advantage Regularization","zh_title":"通过混合模式优势正则化缓解大型推理模型中的事实性幻觉","primary_category":"cs.CL","date":"2026-07-07","score":0,"bucket":"other","tags":["事实性幻觉","强化学习","推理模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.05861","has_summary":false},{"id":"2607.03091","title":"Silicon Sampling via Cross-Survey Transfer","zh_title":"基于跨调查迁移的硅采样","primary_category":"cs.AI","date":"2026-07-03","score":10,"bucket":"selected","tags":["LLM人类仿真","调查方法","算法保真度"],"rubric_hits":["A1","A2","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2607.03091","has_summary":true},{"id":"2607.05440","title":"Retrieval over Reasoning: A Cost-Controlled Benchmark of Language Models for Energy-Retrofit Recommendation","zh_title":"检索优于推理：面向建筑节能改造推荐的语言模型成本控制基准测试","primary_category":"econ.EM","date":"2026-07-03","score":0,"bucket":"other","tags":["建筑节能","LLM基准测试","多标签分类"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2607.05440","has_summary":false},{"id":"2607.00551","title":"Talking Politics with Artificial Intelligence","zh_title":"与人工智能谈论政治","primary_category":"econ.GN","date":"2026-07-01","score":5,"bucket":"other","tags":["政治表达","LLM对话分析","用户行为"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2607.00551","has_summary":true},{"id":"2606.30372","title":"Using Large Language Models as Low-Cost Statistical Estimators for Human-Response Data","zh_title":"使用大语言模型作为人类响应数据的低成本统计估计器","primary_category":"cs.AI","date":"2026-06-29","score":7,"bucket":"pending","tags":["LLM仿真","统计估计","风险等价性"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2606.30372","has_summary":true},{"id":"2606.30986","title":"The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows","zh_title":"智能体AI的组织行为：人-智能体工作流中的集体智能","primary_category":"cs.CY","date":"2026-06-29","score":5,"bucket":"other","tags":["组织行为模拟","多智能体系统","集体智能"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.30986","has_summary":true},{"id":"2606.30987","title":"Measuring Judgment Quality in Natural-Language Explanations: Evidence from Forecasting Tournaments","zh_title":"衡量自然语言解释中的判断质量：来自预测锦标赛的证据","primary_category":"cs.CL","date":"2026-06-29","score":0,"bucket":"other","tags":["LLM评分","预测锦标赛","解释质量"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.30987","has_summary":false},{"id":"2606.30473","title":"Field Order Should Not Matter: Permutation-Invariant Embedding Model Fine-Tuning for Structured Metadata Retrieval","zh_title":"字段顺序不应重要：面向结构化元数据检索的置换不变嵌入模型微调","primary_category":"cs.CL","date":"2026-06-29","score":0,"bucket":"other","tags":["信息检索","嵌入模型","元数据"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.30473","has_summary":false},{"id":"2606.28978","title":"Can LLMs Hire Fairly? Racial Bias in Resume Screening","zh_title":"大语言模型能公平招聘吗？简历筛选中的种族偏见","primary_category":"cs.CL","date":"2026-06-27","score":9,"bucket":"selected","tags":["LLM仿真","招聘歧视","算法偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.28978","has_summary":true},{"id":"2606.28770","title":"Mechanistic Personality Analysis of LLMs Steering Personality via Latent Feature Interventions","zh_title":"大语言模型的机制性人格分析：通过潜在特征干预操控人格","primary_category":"cs.AI","date":"2026-06-27","score":5,"bucket":"other","tags":["LLM人格","机制可解释性","稀疏自编码器"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2606.28770","has_summary":true},{"id":"2606.26883","title":"EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents","zh_title":"EconSimulacra：基于LLM智能体的社会经济系统数字孪生平台","primary_category":"cs.DL","date":"2026-06-25","score":5,"bucket":"other","tags":["LLM智能体","社会经济模拟","跨域交互"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.26883","has_summary":true},{"id":"2606.25484","title":"From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities","zh_title":"从因果发现到实施：一个面向29个德国城市电动滑板车移动枢纽规划的智能体AI框架","primary_category":"cs.CY","date":"2026-06-24","score":0,"bucket":"other","tags":["智能体AI","因果发现","城市规划"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.25484","has_summary":false},{"id":"2606.22797","title":"Measuring Behavior Portability in Large Language Models","zh_title":"测量大语言模型的行为可移植性","primary_category":"cs.AI","date":"2026-06-22","score":7,"bucket":"pending","tags":["行为可移植性","LLM决策","仿真评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2606.22797","has_summary":true},{"id":"2606.22974","title":"When Preferences Fail to Become Incentives: A Utility-Behavior Gap in Large Language Models","zh_title":"当偏好未能成为激励：大语言模型中的效用-行为差距","primary_category":"cs.AI","date":"2026-06-22","score":5,"bucket":"other","tags":["LLM偏好测量","效用-行为差距","模型行为一致性"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2606.22974","has_summary":true},{"id":"2606.23633","title":"AI Exposure Scores: what they measure, what they miss, and what comes next","zh_title":"AI暴露分数：它们测量什么、遗漏什么以及下一步","primary_category":"cs.AI","date":"2026-06-22","score":0,"bucket":"other","tags":["AI暴露分数","未来工作","政策研究"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.23633","has_summary":false},{"id":"2606.22337","title":"Theorist Toolbox: Tools for Agent Based LLM-assisted economic theory Research","zh_title":"理论家工具箱：基于智能体的LLM辅助经济理论研究工具","primary_category":"econ.TH","date":"2026-06-21","score":0,"bucket":"other","tags":["多智能体系统","经济理论","自动验证"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.22337","has_summary":false},{"id":"2606.22385","title":"MetaPS: Adaptive Programmatic Strategy Selection for Market Agents","zh_title":"MetaPS：面向市场智能体的自适应程序化策略选择","primary_category":"cs.AI","date":"2026-06-21","score":0,"bucket":"other","tags":["多智能体系统","市场仿真","策略选择"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.22385","has_summary":false},{"id":"2606.21820","title":"Generating Public Health Responses using Survey-Augmented Large Language Models","zh_title":"使用调查增强的大语言模型生成公共卫生响应","primary_category":"cs.SI","date":"2026-06-20","score":9,"bucket":"selected","tags":["LLM仿真","调查数据合成","公共卫生决策"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2606.21820","has_summary":true},{"id":"2606.19904","title":"Toward Temporal Realism in City-Scale Crisis Response Simulation using LLM Agents","zh_title":"面向城市规模危机响应模拟中时序逼真度的LLM智能体研究","primary_category":"cs.SI","date":"2026-06-18","score":8,"bucket":"selected","tags":["LLM仿真","人类行为对照","危机响应"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.19904","has_summary":true},{"id":"2606.19336","title":"Learning User Simulators with Turing Rewards","zh_title":"用图灵奖励学习用户模拟器","primary_category":"cs.CL","date":"2026-06-17","score":8,"bucket":"selected","tags":["用户模拟","强化学习","人类数据对照"],"rubric_hits":["A1","A4","B1"],"abs_url":"https://arxiv.org/abs/2606.19336","has_summary":true},{"id":"2606.18709","title":"LLMs Struggle to Measure What Distinguishes Students of Different Proficiency Levels: A Study of Item Discrimination in Reading Comprehension Assessment","zh_title":"大语言模型难以衡量区分不同水平学生的题目特征：阅读理解评估中题目区分度的研究","primary_category":"cs.CL","date":"2026-06-17","score":5,"bucket":"other","tags":["LLM心理测量","题目区分度","合成学生回答"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2606.18709","has_summary":true},{"id":"2606.17657","title":"Using Cognitive Models to Improve Language Model Simulation of Human Persuasion Games","zh_title":"利用认知模型改进语言模型对人类说服博弈的仿真","primary_category":"cs.AI","date":"2026-06-16","score":9,"bucket":"selected","tags":["LLM人类仿真","认知模型","说服博弈"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.17657","has_summary":true},{"id":"2606.18005","title":"LLM Consumer Behavior Theory: Foundations of a Novel Research Field","zh_title":"LLM消费者行为理论：一个新兴研究领域的基础","primary_category":"cs.AI","date":"2026-06-16","score":7,"bucket":"pending","tags":["LLM仿真","消费者行为","经济学理论"],"rubric_hits":["A1","A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2606.18005","has_summary":true},{"id":"2606.17165","title":"Statistical Foundations of LLM-based A/B Testing: A Surrogacy Framework for Human Causal Inference","zh_title":"基于大语言模型的A/B测试统计基础：面向人类因果推断的替代框架","primary_category":"stat.ME","date":"2026-06-15","score":10,"bucket":"selected","tags":["LLM仿真","A/B测试","因果推断"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3","B4"],"abs_url":"https://arxiv.org/abs/2606.17165","has_summary":true},{"id":"2606.15031","title":"Partial Identification from LLM Prompts","zh_title":"基于大语言模型提示的部分识别","primary_category":"econ.EM","date":"2026-06-13","score":0,"bucket":"other","tags":["部分识别","计量经济学","LLM分类器"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.15031","has_summary":false},{"id":"2606.14113","title":"Simulating Students' Java Programming Errors with Large Language Models","zh_title":"用大语言模型模拟学生的Java编程错误","primary_category":"cs.SE","date":"2026-06-12","score":7,"bucket":"pending","tags":["LLM仿真","编程教育","人类数据对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2606.14113","has_summary":true},{"id":"2606.12848","title":"(Human) Attention Is (Still) All You Need: Human oversight makes AI-assisted social science reliable","zh_title":"（人类）注意力（仍然）是你所需：人类监督使AI辅助社会科学可靠","primary_category":"cs.AI","date":"2026-06-11","score":7,"bucket":"pending","tags":["人机协作","可靠性评估","社会科学研究"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2606.12848","has_summary":true},{"id":"2606.12830","title":"Perceive, Interact, Reason: Building Tool-Augmented Visual Agents for Spatial Reasoning","zh_title":"感知、交互、推理：构建面向空间推理的工具增强视觉智能体","primary_category":"cs.CV","date":"2026-06-11","score":0,"bucket":"other","tags":["视觉智能体","空间推理","工具增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2606.12830","has_summary":false},{"id":"2606.12972","title":"From Prompts to Preferences: An Open-Source Platform for Generative AI-Enhanced Conjoint Analysis","zh_title":"从提示到偏好：一个用于生成式AI增强联合分析的开源平台","primary_category":"cs.HC","date":"2026-06-11","score":0,"bucket":"other","tags":["联合分析","生成式AI","护理机器人"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2606.12972","has_summary":false},{"id":"2606.12369","title":"Should LLM Agents Decide in Social Simulations? Comparing Finite-State and LLM-Based Decision Policies","zh_title":"LLM代理应否在社会模拟中做决策？比较有限状态与基于LLM的决策策略","primary_category":"cs.CY","date":"2026-06-10","score":5,"bucket":"other","tags":["社会模拟","LLM代理","决策策略"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.12369","has_summary":true},{"id":"2606.10989","title":"Null-Space Constrained Low-Rank Adaptation for Response-Specified Large Language Model Unlearning","zh_title":"基于零空间约束低秩适应的响应指定大语言模型遗忘","primary_category":"cs.AI","date":"2026-06-09","score":0,"bucket":"other","tags":["LLM遗忘","低秩适应","模型安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.10989","has_summary":false},{"id":"2606.09198","title":"MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation","zh_title":"MASS：面向社会科学深度研究的记忆增强社会模拟","primary_category":"cs.AI","date":"2026-06-08","score":5,"bucket":"other","tags":["社会模拟","LLM智能体","论文生成"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.09198","has_summary":true},{"id":"2606.08853","title":"AI-Assisted Variance Reduction in Randomized Experiments","zh_title":"随机实验中AI辅助的方差缩减","primary_category":"econ.EM","date":"2026-06-07","score":8,"bucket":"selected","tags":["LLM仿真","实验设计","方差缩减"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2606.08853","has_summary":true},{"id":"2606.06936","title":"Personality Anchoring for Social Simulation: Linking Personality, Social Behavior, and Interaction Success with LLM Agents","zh_title":"社会模拟中的人格锚定：将人格、社会行为与互动成功与LLM智能体关联","primary_category":"cs.HC","date":"2026-06-05","score":5,"bucket":"other","tags":["社会模拟","人格锚定","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.06936","has_summary":true},{"id":"2606.06089","title":"Leveraging LLMs for Unstructured Claims Data Analysis","zh_title":"利用大语言模型进行非结构化理赔数据分析","primary_category":"q-fin.MF","date":"2026-06-04","score":0,"bucket":"other","tags":["精算数据提取","NLP信息抽取","保险理赔"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.06089","has_summary":false},{"id":"2606.05330","title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","zh_title":"基于概率信念追踪的多轮人类可说服性模型","primary_category":"cs.CL","date":"2026-06-03","score":9,"bucket":"selected","tags":["LLM仿真","信念动态","人类数据对照"],"rubric_hits":["A1","A2","A4","B1","B3","B4"],"abs_url":"https://arxiv.org/abs/2606.05330","has_summary":true},{"id":"2606.04978","title":"Probing Outcome-Level Resemblance and Mechanism-Level Alignment in LLM Risk Decisions: Evidence from the St. Petersburg Game","zh_title":"探究大语言模型风险决策中的结果层相似与机制层对齐：来自圣彼得堡博弈的证据","primary_category":"cs.CL","date":"2026-06-03","score":9,"bucket":"selected","tags":["LLM仿真","风险决策","机制对齐"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2606.04978","has_summary":true},{"id":"2606.03030","title":"Do Matching Mechanisms Work with LLM Agents?","zh_title":"匹配机制在LLM代理市场中是否有效？","primary_category":"cs.GT","date":"2026-06-02","score":9,"bucket":"selected","tags":["LLM代理","市场匹配","人类行为对照"],"rubric_hits":["A3","A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2606.03030","has_summary":true},{"id":"2606.03137","title":"Think-Before-Speak: From Internal Evaluation to Public Expression in Multi-Agent Social Simulation","zh_title":"先想后说：多智能体社会模拟中从内部评估到公开表达","primary_category":"cs.AI","date":"2026-06-02","score":5,"bucket":"other","tags":["多智能体模拟","社会仿真","意见动态"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.03137","has_summary":true},{"id":"2606.03763","title":"Merit or networks? What decides where research is published","zh_title":"功绩还是关系网？什么决定研究在哪里发表","primary_category":"econ.GN","date":"2026-06-02","score":0,"bucket":"other","tags":["科学计量学","LLM评估","学术发表"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.03763","has_summary":false},{"id":"2606.02741","title":"Greener Than Humans? Environmental Attitudes in Large Language Models","zh_title":"比人类更环保？大语言模型中的环境态度","primary_category":"cs.CL","date":"2026-06-01","score":8,"bucket":"selected","tags":["LLM仿真","环境态度","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2606.02741","has_summary":true},{"id":"2606.01199","title":"Can LLM Agents Sustain Long-Horizon Organizational Dynamics?","zh_title":"LLM智能体能维持长期组织动态吗？","primary_category":"cs.AI","date":"2026-05-31","score":5,"bucket":"other","tags":["LLM智能体","组织模拟","社会仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2606.01199","has_summary":true},{"id":"2606.00476","title":"Doing What They Say, Not What They Reason: Locating the Faithfulness Gap in LLM Agents","zh_title":"言行不一：定位LLM智能体的保真度差距","primary_category":"cs.AI","date":"2026-05-30","score":7,"bucket":"pending","tags":["过程保真度","LLM智能体","可靠性评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2606.00476","has_summary":true},{"id":"2606.02632","title":"Position: Prioritize Identifying Structure, Not Complex Models, for Scientific Discovery","zh_title":"立场：优先识别结构而非复杂模型以促进科学发现","primary_category":"stat.ML","date":"2026-05-30","score":0,"bucket":"other","tags":["科学发现","方法论批判","机械学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2606.02632","has_summary":false},{"id":"2605.30911","title":"What Makes LVLMs Hallucinate Less? Unveiling the Architectural Factors Behind Hallucination Robustness","zh_title":"什么让大型视觉语言模型减少幻觉？揭示幻觉鲁棒性背后的架构因素","primary_category":"cs.CV","date":"2026-05-29","score":0,"bucket":"other","tags":["幻觉","视觉语言模型","架构分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.30911","has_summary":false},{"id":"2605.30036","title":"Teaching Values to Machines: Simulating Human-Like Behavior in LLMs","zh_title":"向机器传授价值观：在LLM中模拟类人行为","primary_category":"cs.AI","date":"2026-05-28","score":9,"bucket":"selected","tags":["LLM仿真","价值观诱导","人类数据对照"],"rubric_hits":["A1","A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2605.30036","has_summary":true},{"id":"2605.30258","title":"EASE Configuration Facilitates A Reproducible Science of LLM Social Simulations","zh_title":"EASE配置促进可复现的LLM社会模拟科学","primary_category":"cs.MA","date":"2026-05-28","score":7,"bucket":"pending","tags":["LLM社会模拟","多智能体仿真","可复现性"],"rubric_hits":["A3","D3"],"abs_url":"https://arxiv.org/abs/2605.30258","has_summary":true},{"id":"2605.26437","title":"Divergent Minds, Convergent Baselines: A Bounded-Rationality Account of LLM-Human Strategic Behaviour","zh_title":"分歧思维，收敛基线：LLM与人类战略行为的有界理性解释","primary_category":"econ.GN","date":"2026-05-26","score":9,"bucket":"selected","tags":["LLM仿真","有界理性","行为博弈"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2605.26437","has_summary":true},{"id":"2605.26662","title":"AI evaluation may bias perceptions: The importance of context in interpreting academic writing","zh_title":"AI评估可能产生偏见：解读学术写作时语境的重要性","primary_category":"cs.CL","date":"2026-05-26","score":0,"bucket":"other","tags":["AI检测偏差","学术写作分析","方法评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.26662","has_summary":false},{"id":"2605.25680","title":"Simulating Human Memory with Language Models","zh_title":"用语言模型模拟人类记忆","primary_category":"cs.CL","date":"2026-05-25","score":9,"bucket":"selected","tags":["人类仿真","记忆实验","可靠性评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.25680","has_summary":true},{"id":"2605.25505","title":"Generative AI impacts on intra-urban inequality and skill premium in Beijing","zh_title":"生成式AI对北京城市内部不平等与技能溢价的影响","primary_category":"cs.CY","date":"2026-05-25","score":0,"bucket":"other","tags":["生成式AI","城市不平等","技能溢价"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.25505","has_summary":false},{"id":"2605.24319","title":"Omissive Bias in Religious Representation: Benchmarking LLM Answers to Everyday Ethical Decision-making","zh_title":"宗教表征中的遗漏偏差：基准测试LLM对日常伦理决策的回答","primary_category":"cs.LG","date":"2026-05-23","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","遗漏偏差"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.24319","has_summary":true},{"id":"2605.23783","title":"Benchmarking LLMs for Community Governance Simulation with Life-history Narratives","zh_title":"基于生活史叙事的社区治理仿真大语言模型基准测试","primary_category":"cs.CY","date":"2026-05-22","score":9,"bucket":"selected","tags":["LLM人类仿真","社区治理","政策评估"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2605.23783","has_summary":true},{"id":"2605.23867","title":"Human Decision-Making with Persuasive and Narrative LLM Explanations","zh_title":"说服性与叙事性LLM解释对人类决策的影响","primary_category":"cs.HC","date":"2026-05-22","score":7,"bucket":"pending","tags":["人类决策实验","AI解释","行为实验"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2605.23867","has_summary":false},{"id":"2605.23159","title":"Generative AI and the Reorganization of Labor Demand","zh_title":"生成式人工智能与劳动力需求的重组","primary_category":"econ.GN","date":"2026-05-22","score":0,"bucket":"other","tags":["劳动力市场","AI暴露度","岗位分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.23159","has_summary":false},{"id":"2605.22095","title":"Not Yet: Humans Outperform LLMs in a Colonel Blotto Tournament","zh_title":"尚未：人类在Colonel Blotto锦标赛中胜过LLM","primary_category":"econ.GN","date":"2026-05-21","score":8,"bucket":"selected","tags":["LLM仿真","行为博弈","人类对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2605.22095","has_summary":true},{"id":"2605.21401","title":"Open-source LLMs administer maximum electric shocks in a Milgram-like obedience experiment","zh_title":"开源大语言模型在类米尔格拉姆服从实验中施加最大电击","primary_category":"cs.CY","date":"2026-05-20","score":9,"bucket":"selected","tags":["LLM仿真","服从实验","人类行为对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.21401","has_summary":true},{"id":"2605.27419","title":"APS: Bias-Controlled Adaptive Prototype Simulation for Population-Scale LLM Agents","zh_title":"面向大规模LLM智能体的偏差控制自适应原型仿真","primary_category":"cs.MA","date":"2026-05-19","score":7,"bucket":"pending","tags":["社会模拟","偏差控制","舆论仿真"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2605.27419","has_summary":true},{"id":"2605.19351","title":"PAVE: A Cognitive Architecture for Legitimate Violation in Generative Agent Societies","zh_title":"PAVE：生成式智能体社会中正当违规的认知架构","primary_category":"cs.MA","date":"2026-05-19","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","认知架构"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.19351","has_summary":true},{"id":"2605.22855","title":"PrefBench: Evaluating Zero-Shot LLM Agents in Hidden-Preference Personalized Pricing Negotiations","zh_title":"PrefBench：评估隐藏偏好个性化定价谈判中的零样本LLM智能体","primary_category":"cs.GT","date":"2026-05-19","score":5,"bucket":"other","tags":["LLM智能体","定价谈判","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.22855","has_summary":true},{"id":"2605.18311","title":"Distorted Perspectives of LLM-Simulated Preferences: Can AI Mislead Design?","zh_title":"LLM模拟偏好的扭曲视角：AI会误导设计吗？","primary_category":"cs.HC","date":"2026-05-18","score":9,"bucket":"selected","tags":["LLM仿真","人类数据对照","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.18311","has_summary":true},{"id":"2605.18357","title":"Engagement vs. Commitment: The Economic Trade-Offs of Polarizing News Content","zh_title":"参与与承诺：极化新闻内容的经济权衡","primary_category":"econ.GN","date":"2026-05-18","score":0,"bucket":"other","tags":["新闻极化","LLM测量","经济权衡"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.18357","has_summary":false},{"id":"2605.17979","title":"Comment on Scientific production in the era of large language models","zh_title":"评《大语言模型时代的科学生产》","primary_category":"econ.EM","date":"2026-05-18","score":0,"bucket":"other","tags":["方法论批评","因果推断","LLM检测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.17979","has_summary":false},{"id":"2605.18890","title":"Stop Drawing Scientific Claims from LLM Social Simulations Without Robustness Audits","zh_title":"停止从LLM社会仿真中得出科学结论而不进行稳健性审计","primary_category":"physics.soc-ph","date":"2026-05-17","score":9,"bucket":"selected","tags":["LLM社会仿真","稳健性审计","人类行为对照"],"rubric_hits":["A3","A2","B4","B1"],"abs_url":"https://arxiv.org/abs/2605.18890","has_summary":true},{"id":"2605.17086","title":"Global Automation Atlas","zh_title":"全球自动化图谱","primary_category":"econ.GN","date":"2026-05-16","score":5,"bucket":"other","tags":["LLM标注","自动化暴露","劳动经济学"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2605.17086","has_summary":false},{"id":"2605.16193","title":"Improving Cross-Cultural Survey Simulation with Calibrated Value Personas","zh_title":"基于校准价值观人格的跨文化调查仿真改进","primary_category":"cs.CL","date":"2026-05-15","score":9,"bucket":"selected","tags":["LLM人类仿真","跨文化调查","价值观校准"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2605.16193","has_summary":true},{"id":"2606.12433","title":"Marginal Alignment Does Not Guarantee Joint-Distribution Fidelity: An Official-Reference Audit of Nemotron-Personas-Korea with Cross-Locale Replication","zh_title":"边际对齐不保证联合分布保真度：对Nemotron-Personas-Korea的官方参考审计及跨地区复现","primary_category":"cs.CY","date":"2026-05-15","score":5,"bucket":"other","tags":["合成数据审计","人物数据集","联合分布保真度"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2606.12433","has_summary":true},{"id":"2605.15734","title":"Can We Trust AI-Inferred User States. A Psychometric Framework for Validating the Reliability of Users States Classification by LLMs in Operational Environments","zh_title":"我们能信任AI推断的用户状态吗？一个验证LLM在操作环境中用户状态分类信度的心理计量框架","primary_category":"cs.AI","date":"2026-05-15","score":5,"bucket":"other","tags":["LLM测量信度","心理计量验证","用户状态推断"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2605.15734","has_summary":true},{"id":"2605.12898","title":"When Do LLMs Generate Realistic Social Networks? A Multi-Dimensional Study of Culture, Language, Scale, and Method","zh_title":"大语言模型何时生成真实的社交网络？一项关于文化、语言、规模和方法的多维研究","primary_category":"cs.SI","date":"2026-05-13","score":9,"bucket":"selected","tags":["LLM仿真","社交网络生成","人类数据对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2605.12898","has_summary":true},{"id":"2605.13307","title":"PRISM-X: Experiments on Personalised Fine-Tuning with Human and Simulated Users","zh_title":"PRISM-X：基于人类与模拟用户的个性化微调实验","primary_category":"cs.CL","date":"2026-05-13","score":9,"bucket":"selected","tags":["LLM仿真","人类数据对照","个性化评估"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.13307","has_summary":true},{"id":"2605.13725","title":"ScioMind: Cognitively Grounded Multi-Agent Social Simulation with Anchoring-Based Belief Dynamics and Dynamic Profiles","zh_title":"ScioMind：基于锚定信念动态和动态画像的认知基础多智能体社会模拟","primary_category":"cs.AI","date":"2026-05-13","score":6,"bucket":"other","tags":["社会模拟","多智能体","认知基础"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.13725","has_summary":true},{"id":"2605.12147","title":"PrivacySIM: Evaluating LLM Simulation of User Privacy Behavior","zh_title":"PrivacySIM：评估大语言模型对用户隐私行为的仿真","primary_category":"cs.CR","date":"2026-05-12","score":9,"bucket":"selected","tags":["LLM仿真","隐私行为","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.12147","has_summary":true},{"id":"2606.18263","title":"How Well Do Large Language Models Capture Human Personality?","zh_title":"大语言模型捕捉人类人格的效果如何？","primary_category":"cs.HC","date":"2026-05-12","score":8,"bucket":"selected","tags":["LLM人格仿真","仿真保真度","人格坍缩"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2606.18263","has_summary":true},{"id":"2605.11404","title":"Attributing Emergence in Million-Agent Systems","zh_title":"百万智能体系统中的涌现归因","primary_category":"cs.AI","date":"2026-05-12","score":7,"bucket":"pending","tags":["LLM社会模拟","涌现归因","大规模多智能体"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.11404","has_summary":true},{"id":"2605.12824","title":"Mechanism Plausibility in Generative Agent-Based Modeling","zh_title":"生成式智能体建模中的机制合理性","primary_category":"cs.MA","date":"2026-05-12","score":5,"bucket":"other","tags":["LLM智能体建模","社会模拟","机制解释"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.12824","has_summary":true},{"id":"2605.12618","title":"Career Mobility of Planning Alumni in the United States: Evidence from Professional Profile Data using Large Language Models","zh_title":"美国规划校友的职业流动性：基于大语言模型的专业档案数据证据","primary_category":"cs.CY","date":"2026-05-12","score":0,"bucket":"other","tags":["职业流动性","信息抽取","规划教育"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.12618","has_summary":false},{"id":"2605.10659","title":"When Can Digital Personas Reliably Approximate Human Survey Findings?","zh_title":"数字人何时能可靠近似人类调查发现？","primary_category":"cs.CL","date":"2026-05-11","score":10,"bucket":"selected","tags":["LLM仿真","调查方法","算法保真度"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.10659","has_summary":true},{"id":"2605.22841","title":"Strategic Coercion Within Alliances: The Greenland Sovereignty Game as an AI Stress Test","zh_title":"联盟内的战略胁迫：格陵兰主权博弈作为AI压力测试","primary_category":"physics.soc-ph","date":"2026-05-11","score":7,"bucket":"pending","tags":["LLM地缘政治模拟","多智能体博弈","逆博弈论"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2605.22841","has_summary":true},{"id":"2605.10505","title":"A Theory of Multilevel Interactive Equilibrium in NeuroAI","zh_title":"神经AI中多层次交互均衡理论","primary_category":"cs.NE","date":"2026-05-11","score":0,"bucket":"other","tags":["博弈论","多智能体系统","神经AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.10505","has_summary":false},{"id":"2605.10831","title":"SLIM: Sparse Latent Steering for Interpretable and Property-Directed LLM-Based Molecular Editing","zh_title":"SLIM：面向可解释与属性导向的基于大语言模型的分子编辑的稀疏潜在操控","primary_category":"cs.LG","date":"2026-05-11","score":0,"bucket":"other","tags":["分子编辑","稀疏自编码器","属性控制"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.10831","has_summary":false},{"id":"2607.20429","title":"More Is Not More: What Matters for Diversity in LLM Opinions?","zh_title":"越多并非越好：什么因素影响LLM意见的多样性？","primary_category":"cs.CL","date":"2026-05-10","score":9,"bucket":"selected","tags":["LLM人类仿真","意见多样性","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2607.20429","has_summary":true},{"id":"2606.14715","title":"MiroBench: Benchmarking Realism in Agentic Simulation of Real-world Discussions","zh_title":"MiroBench：基准测试真实世界讨论的智能体仿真真实性","primary_category":"cs.MA","date":"2026-05-10","score":9,"bucket":"selected","tags":["LLM仿真","社会模拟","真实性评估"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2606.14715","has_summary":true},{"id":"2605.08837","title":"The Grounding Gap: How LLMs Anchor the Meaning of Abstract Concepts Differently from Humans","zh_title":"接地差距：大语言模型如何以不同于人类的方式锚定抽象概念的意义","primary_category":"cs.CL","date":"2026-05-09","score":7,"bucket":"pending","tags":["LLM仿真","认知实验复现","概念接地"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.08837","has_summary":true},{"id":"2605.08802","title":"CoLVR: Enhancing Exploratory Latent Visual Reasoning via Contrastive Optimization","zh_title":"CoLVR：通过对比优化增强探索性潜在视觉推理","primary_category":"cs.CV","date":"2026-05-09","score":0,"bucket":"other","tags":["视觉推理","多模态大模型","对比学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.08802","has_summary":false},{"id":"2605.07692","title":"GASim: A Graph-Accelerated Hybrid Framework for Social Simulation","zh_title":"GASim：一种图加速的混合社会仿真框架","primary_category":"cs.AI","date":"2026-05-08","score":6,"bucket":"other","tags":["社会模拟","LLM智能体","图加速"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.07692","has_summary":true},{"id":"2605.05578","title":"Artificial Aesthetics: The Implicit Economics of Valuing AI-Generated Text","zh_title":"人工美学：评估AI生成文本的隐含经济学","primary_category":"econ.GN","date":"2026-05-07","score":0,"bucket":"other","tags":["支付意愿","AI文本评估","行为经济学"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.05578","has_summary":false},{"id":"2605.06987","title":"Response Time Enhances Alignment with Heterogeneous Preferences","zh_title":"响应时间增强异质性偏好对齐","primary_category":"cs.LG","date":"2026-05-07","score":0,"bucket":"other","tags":["偏好对齐","响应时间","异质性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.06987","has_summary":false},{"id":"2605.18781","title":"Can LLMs Emulate Human Belief Dynamics?","zh_title":"大语言模型能模拟人类信念动态吗？","primary_category":"cs.SI","date":"2026-05-05","score":10,"bucket":"selected","tags":["LLM仿真","信念动态","人类数据对照"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.18781","has_summary":true},{"id":"2605.03604","title":"Multi-Agent Strategic Games with LLMs","zh_title":"基于大语言模型的多智能体战略博弈研究","primary_category":"cs.GT","date":"2026-05-05","score":9,"bucket":"selected","tags":["LLM仿真","战略博弈","国际关系"],"rubric_hits":["A1","A3","B2","B3"],"abs_url":"https://arxiv.org/abs/2605.03604","has_summary":true},{"id":"2605.03287","title":"Attention: What Prevents Young Adults from Speaking Up Against Cyberbullying in an LLM-Powered Social Media Simulation","zh_title":"注意力：是什么阻止年轻人在LLM驱动的社交媒体仿真中公开反对网络欺凌","primary_category":"cs.HC","date":"2026-05-05","score":5,"bucket":"other","tags":["多智能体仿真","网络欺凌干预","人机交互"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.03287","has_summary":true},{"id":"2605.04029","title":"Stayin' Aligned Over Time: Towards Longitudinal Human-LLM Alignment via Contextual Reflection and Privacy-Preserving Behavioral Data","zh_title":"随时间保持对齐：通过情境反思和隐私保护行为数据实现纵向人-LLM对齐","primary_category":"cs.HC","date":"2026-05-05","score":5,"bucket":"other","tags":["人机对齐","纵向研究","偏好测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2605.04029","has_summary":true},{"id":"2605.02598","title":"What Jobs Can AI Learn? Measuring Exposure by Reinforcement Learning","zh_title":"AI能学会哪些工作？用强化学习衡量职业暴露度","primary_category":"econ.GN","date":"2026-05-04","score":0,"bucket":"other","tags":["AI暴露度","职业分类","强化学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2605.02598","has_summary":false},{"id":"2606.11217","title":"Preregistration for Experiments with AI Agents","zh_title":"AI代理实验的预注册","primary_category":"cs.CY","date":"2026-05-03","score":8,"bucket":"selected","tags":["AI代理实验","预注册","方法论"],"rubric_hits":["A4","B4"],"abs_url":"https://arxiv.org/abs/2606.11217","has_summary":true},{"id":"2605.01311","title":"The Partial Testimony of Logs: Evaluation of Language Model Generation under Confounded Model Choice","zh_title":"日志的部分证言：混杂模型选择下的语言模型生成评估","primary_category":"cs.LG","date":"2026-05-02","score":0,"bucket":"other","tags":["离线评估","因果推断","语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.01311","has_summary":false},{"id":"2604.23575","title":"The Collapse of Heterogeneity in Silicon Philosophers","zh_title":"硅基哲学家的异质性坍塌","primary_category":"cs.CY","date":"2026-04-26","score":9,"bucket":"selected","tags":["LLM人类仿真","异质性评估","哲学观点复现"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.23575","has_summary":true},{"id":"2604.23897","title":"MarketBench: Evaluating AI Agents as Market Participants","zh_title":"MarketBench：评估AI智能体作为市场参与者","primary_category":"cs.AI","date":"2026-04-26","score":5,"bucket":"other","tags":["AI智能体","市场模拟","基准测试"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.23897","has_summary":true},{"id":"2605.27401","title":"Using Zero-Shot LLM-Generated Survey Data for Geographically Explicit Population Synthesis","zh_title":"使用零样本LLM生成调查数据进行地理显式人口合成","primary_category":"cs.CY","date":"2026-04-23","score":8,"bucket":"selected","tags":["LLM仿真","人口合成","健康调查"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2605.27401","has_summary":true},{"id":"2604.21334","title":"Ideological Bias in LLMs' Economic Causal Reasoning","zh_title":"大语言模型经济因果推理中的意识形态偏差","primary_category":"cs.AI","date":"2026-04-23","score":6,"bucket":"other","tags":["意识形态偏差","经济因果推理","LLM评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2604.21334","has_summary":true},{"id":"2604.20652","title":"Large Language Models Outperform Humans in Fraud Detection and Resistance to Motivated Investor Pressure","zh_title":"大语言模型在欺诈检测和抵制动机性投资者压力方面优于人类","primary_category":"cs.AI","date":"2026-04-22","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","欺诈检测"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.20652","has_summary":true},{"id":"2604.19925","title":"Behavioral Transfer in AI Agents: Evidence and Privacy Implications","zh_title":"AI代理中的行为转移：证据与隐私影响","primary_category":"econ.GN","date":"2026-04-21","score":8,"bucket":"selected","tags":["AI代理","行为转移","人类仿真"],"rubric_hits":["A1","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.19925","has_summary":true},{"id":"2604.19260","title":"Understanding the Mechanism of Altruism in Large Language Models","zh_title":"理解大语言模型中利他行为的机制","primary_category":"econ.GN","date":"2026-04-21","score":7,"bucket":"pending","tags":["LLM仿真","利他行为","机制可解释性"],"rubric_hits":["A1","A5","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.19260","has_summary":true},{"id":"2604.18373","title":"Dissecting AI Trading: Behavioral Finance and Market Bubbles","zh_title":"剖析AI交易：行为金融与市场泡沫","primary_category":"econ.GN","date":"2026-04-20","score":9,"bucket":"selected","tags":["LLM仿真","行为金融","实验市场"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.18373","has_summary":true},{"id":"2604.18011","title":"Topology-Aware LLM-Driven Social Simulation: A Unified Framework for Efficient and Realistic Agent Dynamics","zh_title":"拓扑感知的LLM驱动社会仿真：高效且逼真的智能体动态统一框架","primary_category":"cs.SI","date":"2026-04-20","score":5,"bucket":"other","tags":["社会仿真","多智能体","网络拓扑"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.18011","has_summary":true},{"id":"2604.17774","title":"Prompt Optimization Enables Stable Algorithmic Collusion in LLM Agents","zh_title":"提示优化使LLM智能体实现稳定的算法合谋","primary_category":"cs.AI","date":"2026-04-20","score":5,"bucket":"other","tags":["LLM智能体","市场模拟","算法合谋"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.17774","has_summary":true},{"id":"2604.17267","title":"Rectification Difficulty and Optimal Sample Allocation in LLM-Augmented Surveys","zh_title":"LLM增强调查中的校正难度与最优样本分配","primary_category":"cs.AI","date":"2026-04-19","score":7,"bucket":"pending","tags":["LLM仿真","调查实验","样本分配"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2604.17267","has_summary":true},{"id":"2604.17615","title":"WhatIf: Interactive Exploration of LLM-Powered Social Simulations for Policy Reasoning","zh_title":"WhatIf：用于政策推理的LLM驱动社会模拟的交互式探索","primary_category":"cs.HC","date":"2026-04-19","score":7,"bucket":"pending","tags":["LLM社会模拟","政策评估","交互式系统"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2604.17615","has_summary":true},{"id":"2604.17220","title":"Dynamics of Cognitive Heterogeneity: Investigating Behavioral Biases in Multi-Stage Supply Chains with LLM-Based Simulation","zh_title":"认知异质性动力学：基于LLM仿真的多级供应链行为偏差研究","primary_category":"cs.MA","date":"2026-04-19","score":7,"bucket":"pending","tags":["LLM仿真","供应链行为","认知异质性"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.17220","has_summary":true},{"id":"2605.23920","title":"Artificial Effort","zh_title":"人工努力：大语言模型对实验经济学中真实努力任务的影响","primary_category":"cs.CY","date":"2026-04-17","score":8,"bucket":"selected","tags":["LLM仿真","实验经济学","可靠性评估"],"rubric_hits":["A2","B4"],"abs_url":"https://arxiv.org/abs/2605.23920","has_summary":true},{"id":"2604.14786","title":"CogEvolution: A Human-like Generative Educational Agent to Simulate Student's Cognitive Evolution","zh_title":"CogEvolution：模拟学生认知演化的人类化生成式教育智能体","primary_category":"cs.AI","date":"2026-04-16","score":5,"bucket":"other","tags":["教育智能体","认知演化","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.14786","has_summary":true},{"id":"2604.14575","title":"Generative Augmented Inference of LLM-generated Data for Market Research: Theory and Empirical Evidence","zh_title":"基于LLM生成数据的生成式增强推断用于市场研究：理论与实证","primary_category":"cs.LG","date":"2026-04-16","score":5,"bucket":"other","tags":["LLM辅助推断","市场研究","合成数据"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2604.14575","has_summary":true},{"id":"2604.14467","title":"Who Saw It Coming? Historical Experience and the 2021 Inflation Forecast Failure","zh_title":"谁预见到了？历史经验与2021年通胀预测失败","primary_category":"econ.EM","date":"2026-04-15","score":7,"bucket":"pending","tags":["LLM仿真","通胀预测","经验学习"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2604.14467","has_summary":true},{"id":"2604.14386","title":"Coalition Formation in LLM Agent Networks: Stability Analysis and Convergence Guarantees","zh_title":"LLM智能体网络中的联盟形成：稳定性分析与收敛保证","primary_category":"cs.GT","date":"2026-04-15","score":0,"bucket":"other","tags":["多智能体系统","博弈论","联盟形成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.14386","has_summary":false},{"id":"2604.11312","title":"Network Effects and Agreement Drift in LLM Debates","zh_title":"LLM辩论中的网络效应与意见漂移","primary_category":"cs.SI","date":"2026-04-13","score":5,"bucket":"other","tags":["LLM社会模拟","意见动态","网络效应"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.11312","has_summary":true},{"id":"2604.10834","title":"LLMs for Qualitative Data Analysis Fail on Security-specificComments in Human Experiments","zh_title":"大语言模型在人类实验安全评论的定性数据分析中失效","primary_category":"cs.SE","date":"2026-04-12","score":5,"bucket":"other","tags":["LLM标注","定性数据分析","安全评论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2604.10834","has_summary":true},{"id":"2605.23916","title":"Agent-Facing Information Design in LLM Tool Registries","zh_title":"面向智能体的LLM工具注册信息设计","primary_category":"cs.IR","date":"2026-04-12","score":0,"bucket":"other","tags":["多智能体系统","信息设计","工具注册"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2605.23916","has_summary":false},{"id":"2604.10029","title":"Self-Distilled Reinforcement Learning for Co-Evolving Agentic Recommender Systems","zh_title":"用于协同进化智能体推荐系统的自蒸馏强化学习","primary_category":"cs.IR","date":"2026-04-11","score":0,"bucket":"other","tags":["智能体推荐系统","强化学习","多智能体协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.10029","has_summary":false},{"id":"2604.09502","title":"Strategic Algorithmic Monoculture: Experimental Evidence from Coordination Games","zh_title":"策略性算法单一文化：来自协调博弈的实验证据","primary_category":"cs.AI","date":"2026-04-10","score":9,"bucket":"selected","tags":["LLM仿真","人类行为对照","协调博弈"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2604.09502","has_summary":true},{"id":"2604.16472","title":"Training Language Models for Bilateral Trade with Private Information","zh_title":"训练语言模型进行具有私人信息的双边贸易","primary_category":"cs.GT","date":"2026-04-10","score":5,"bucket":"other","tags":["LLM Agent","双边贸易","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.16472","has_summary":true},{"id":"2604.09855","title":"Instructing LLMs to Negotiate using Reinforcement Learning with Verifiable Rewards","zh_title":"使用可验证奖励的强化学习指导大语言模型进行谈判","primary_category":"cs.AI","date":"2026-04-10","score":0,"bucket":"other","tags":["多智能体系统","强化学习","谈判策略"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.09855","has_summary":false},{"id":"2604.09187","title":"The Geoeconomics of Venture Capital An Economic Complexity Approach to Emerging Technological Sovereignty","zh_title":"风险投资的地缘经济学：新兴技术主权的经济复杂性方法","primary_category":"econ.GN","date":"2026-04-10","score":0,"bucket":"other","tags":["风险投资","经济复杂性","技术分类"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.09187","has_summary":false},{"id":"2604.08678","title":"Scaffolding Human-AI Collaboration: A Field Experiment on Behavioral Protocols and Cognitive Reframing","zh_title":"搭建人机协作框架：关于行为协议与认知重构的现场实验","primary_category":"econ.GN","date":"2026-04-09","score":0,"bucket":"other","tags":["人机协作","现场实验","生成式AI"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.08678","has_summary":false},{"id":"2604.06663","title":"Restoring Heterogeneity in LLM-based Social Simulation: An Audience Segmentation Approach","zh_title":"在基于大语言模型的社会模拟中恢复异质性：一种受众细分方法","primary_category":"cs.CY","date":"2026-04-08","score":9,"bucket":"selected","tags":["LLM人类仿真","受众细分","社会模拟保真度"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2604.06663","has_summary":true},{"id":"2604.16465","title":"Healthcare AI for Automation or Allocation? A Transaction Cost Economics Framework","zh_title":"医疗AI用于自动化还是分配？一个交易成本经济学框架","primary_category":"cs.AI","date":"2026-04-08","score":0,"bucket":"other","tags":["交易成本经济学","职业分析","LLM标注"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.16465","has_summary":false},{"id":"2604.08606","title":"Extrapolating Volition with Recursive Information Markets","zh_title":"用递归信息市场外推意志","primary_category":"cs.GT","date":"2026-04-08","score":0,"bucket":"other","tags":["信息市场","多智能体","AI对齐"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.08606","has_summary":false},{"id":"2604.06860","title":"Personalization as a Game: Equilibrium-Guided Generative Modeling for Physician Behavior in Pharmaceutical Engagement","zh_title":"作为博弈的个性化：制药参与中医生行为的均衡引导生成建模","primary_category":"cs.GT","date":"2026-04-08","score":0,"bucket":"other","tags":["博弈论","生成AI","制药营销"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.06860","has_summary":false},{"id":"2604.05939","title":"Context-Value-Action Architecture for Value-Driven Large Language Model Agents","zh_title":"面向价值驱动大语言模型智能体的情境-价值-行动架构","primary_category":"cs.AI","date":"2026-04-07","score":9,"bucket":"selected","tags":["人类行为仿真","价值驱动智能体","算法保真度"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.05939","has_summary":true},{"id":"2604.05516","title":"Coupling Macro Dynamics and Micro States for Long-Horizon Social Simulation","zh_title":"耦合宏观动态与微观状态的长周期社会模拟","primary_category":"cs.SI","date":"2026-04-07","score":5,"bucket":"other","tags":["社会模拟","舆论动态","LLM agent"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.05516","has_summary":true},{"id":"2604.04464","title":"Bounded by Risk, Not Capability: Quantifying AI Occupational Substitution Rates via a Tech-Risk Dual-Factor Model","zh_title":"受风险而非能力约束：通过技术-风险双因素模型量化AI职业替代率","primary_category":"cs.CY","date":"2026-04-06","score":5,"bucket":"other","tags":["AI职业替代","多智能体仿真","劳动力市场"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2604.04464","has_summary":false},{"id":"2604.03920","title":"From Plausible to Causal: Counterfactual Semantics for Policy Evaluation in Simulated Online Communities","zh_title":"从合理到因果：模拟在线社区中政策评估的反事实语义","primary_category":"cs.CL","date":"2026-04-05","score":7,"bucket":"pending","tags":["LLM社会模拟","政策评估","因果推断"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2604.03920","has_summary":true},{"id":"2605.00841","title":"AI Agents for Sustainable SMEs: A Green ESG Assessment Framework","zh_title":"面向可持续中小企业的AI代理：绿色ESG评估框架","primary_category":"cs.AI","date":"2026-04-05","score":5,"bucket":"other","tags":["AI代理","ESG评估","标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2605.00841","has_summary":true},{"id":"2605.20191","title":"Shiny Stories, Hidden Struggles: Investigating the Representation of Disability Through the Lens of LLMs","zh_title":"光鲜故事，隐藏挣扎：通过LLM视角考察残障表征","primary_category":"cs.CL","date":"2026-04-02","score":9,"bucket":"selected","tags":["LLM仿真","残障表征","偏差评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2605.20191","has_summary":true},{"id":"2604.01520","title":"LLM Agents as Social Scientists: A Human-AI Collaborative Platform for Social Science Automation","zh_title":"作为社会科学家的LLM代理：一个面向社会科学自动化的人机协作平台","primary_category":"cs.AI","date":"2026-04-02","score":9,"bucket":"selected","tags":["LLM人类仿真","社会科学自动化","人机协作"],"rubric_hits":["A1","A3","A5","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2604.01520","has_summary":true},{"id":"2604.01896","title":"Bayesian Elicitation with LLMs: Model Size Helps, Extra \"Reasoning\" Doesn't Always","zh_title":"基于大语言模型的贝叶斯启发：模型规模有益，额外“推理”未必有效","primary_category":"cs.AI","date":"2026-04-02","score":7,"bucket":"pending","tags":["LLM仿真","贝叶斯启发","不确定性校准"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.01896","has_summary":true},{"id":"2604.02403","title":"Measuring What Cannot Be Surveyed: LLMs as Instruments for Latent Cognitive Variables in Labor Economics","zh_title":"测量不可调查之物：LLM作为劳动经济学中潜在认知变量的工具","primary_category":"econ.EM","date":"2026-04-02","score":5,"bucket":"other","tags":["LLM测量工具","劳动经济学","认知变量"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2604.02403","has_summary":true},{"id":"2604.01066","title":"Augmented Human Capital: A Unified Theory and LLM-Based Measurement Framework for Cognitive Factor Decomposition in AI-Augmented Economies","zh_title":"增强型人力资本：AI增强经济中认知因素分解的统一理论与基于LLM的测量框架","primary_category":"econ.GN","date":"2026-04-01","score":5,"bucket":"other","tags":["LLM标注","人力资本","劳动经济学"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2604.01066","has_summary":false},{"id":"2604.01416","title":"Pay-Per-Crawl Pricing for AI: The LM-Tree Agent","zh_title":"AI按次抓取付费定价：LM-Tree智能体","primary_category":"econ.GN","date":"2026-04-01","score":0,"bucket":"other","tags":["AI定价","多智能体系统","内容付费"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2604.01416","has_summary":false},{"id":"2604.01363","title":"Crashing Waves vs. Rising Tides: Findings on AI Automation from Thousands of Worker Evaluations of Labor Market Tasks","zh_title":"惊涛骇浪还是水涨船高：基于数千名劳动者对劳动力市场任务评估的AI自动化发现","primary_category":"cs.AI","date":"2026-04-01","score":0,"bucket":"other","tags":["AI自动化","劳动力市场","任务评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2604.01363","has_summary":false},{"id":"2604.25922","title":"Consciousness with the Serial Numbers Filed Off: Measuring Trained Denial in 115 AI Models","zh_title":"抹去序列号的意识：测量115个AI模型中的训练性否认","primary_category":"cs.CL","date":"2026-04-01","score":0,"bucket":"other","tags":["AI意识","基准测试","对齐失败"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2604.25922","has_summary":false},{"id":"2603.29741","title":"BotVerse: Real-Time Event-Driven Simulation of Social Agents","zh_title":"BotVerse：基于实时事件驱动的社交智能体仿真","primary_category":"cs.SI","date":"2026-03-31","score":5,"bucket":"other","tags":["社会模拟","LLM智能体","虚假信息"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.29741","has_summary":true},{"id":"2603.29121","title":"Economics of Human and AI Collaboration: When is Partial Automation More Attractive than Full Automation?","zh_title":"人机协作经济学：部分自动化何时比完全自动化更具吸引力？","primary_category":"econ.GN","date":"2026-03-31","score":0,"bucket":"other","tags":["自动化经济学","人机协作","任务复杂度"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.29121","has_summary":false},{"id":"2603.27956","title":"Artificial Intelligence in Science: Returns, Reallocation, and Reorganization","zh_title":"科学中的人工智能：回报、资源再配置与重组","primary_category":"physics.soc-ph","date":"2026-03-30","score":0,"bucket":"other","tags":["科学学","AI经济学","科研生产力"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2603.27956","has_summary":false},{"id":"2603.27056","title":"Persona-Based Simulation of Human Opinion at Population Scale","zh_title":"基于人格的群体意见仿真：从社交媒体推断半结构化人格以驱动LLM代理","primary_category":"cs.CY","date":"2026-03-28","score":10,"bucket":"selected","tags":["LLM人类仿真","人格推断","调查方法"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2603.27056","has_summary":true},{"id":"2603.24947","title":"Shopping with a Platform AI Assistant: Who Adopts, When in the Journey, and What For","zh_title":"与平台AI助手购物：谁采用、在旅程的何时、以及用于什么","primary_category":"cs.AI","date":"2026-03-26","score":0,"bucket":"other","tags":["用户行为分析","电子商务","AI助手"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.24947","has_summary":false},{"id":"2603.23884","title":"POSIM: A Multi-Agent Simulation Framework for Social Media Public Opinion Evolution and Governance","zh_title":"POSIM：社交媒体舆论演化与治理的多智能体仿真框架","primary_category":"cs.GL","date":"2026-03-25","score":9,"bucket":"selected","tags":["LLM仿真","舆论演化","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2603.23884","has_summary":true},{"id":"2603.22837","title":"Analysing LLM Persona Generation and Fairness Interpretation in Polarised Geopolitical Contexts","zh_title":"分析极化地缘政治背景下LLM人格生成与公平性解释","primary_category":"cs.CL","date":"2026-03-24","score":5,"bucket":"other","tags":["LLM人格生成","地缘政治偏见","社会模拟"],"rubric_hits":["D2","D3"],"abs_url":"https://arxiv.org/abs/2603.22837","has_summary":true},{"id":"2603.21690","title":"AI Token Futures Market: Commoditization of Compute and Derivatives Contract Design","zh_title":"AI代币期货市场：算力的商品化与衍生品合约设计","primary_category":"cs.AI","date":"2026-03-23","score":0,"bucket":"other","tags":["算力金融化","期货合约设计","蒙特卡洛模拟"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.21690","has_summary":false},{"id":"2603.20678","title":"AI-Driven Multi-Agent Simulation of Stratified Polyamory Systems: A Computational Framework for Optimizing Social Reproductive Efficiency","zh_title":"AI驱动的分层多偶制系统多智能体仿真：优化社会生育效率的计算框架","primary_category":"cs.AI","date":"2026-03-21","score":5,"bucket":"other","tags":["社会模拟","多智能体系统","计算社会科学"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.20678","has_summary":true},{"id":"2603.19791","title":"Text-Based Personas for Simulating User Privacy Decisions","zh_title":"基于文本角色模拟用户隐私决策","primary_category":"cs.CR","date":"2026-03-20","score":9,"bucket":"selected","tags":["隐私决策仿真","合成角色","人类数据对照"],"rubric_hits":["A1","A5","B1","B2"],"abs_url":"https://arxiv.org/abs/2603.19791","has_summary":true},{"id":"2603.19649","title":"PolicySim: An LLM-Based Agent Social Simulation Sandbox for Proactive Policy Optimization","zh_title":"PolicySim：一个基于LLM的智能体社会仿真沙盒，用于主动政策优化","primary_category":"cs.SI","date":"2026-03-20","score":7,"bucket":"pending","tags":["社会仿真","政策评估","LLM智能体"],"rubric_hits":["A3","B2"],"abs_url":"https://arxiv.org/abs/2603.19649","has_summary":true},{"id":"2605.12507","title":"Can LLM Agents Simulate Dynamic Networks? A Case Study on Email Networks with Phishing Synthesis","zh_title":"LLM智能体能模拟动态网络吗？以钓鱼邮件合成为例的邮件网络案例研究","primary_category":"cs.SI","date":"2026-03-20","score":5,"bucket":"other","tags":["LLM智能体","动态网络模拟","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2605.12507","has_summary":true},{"id":"2603.18563","title":"Reasonably reasoning AI agents can avoid game-theoretic failures in zero-shot, provably","zh_title":"合理推理的AI智能体可零样本避免博弈论失败，且可证明","primary_category":"cs.AI","date":"2026-03-19","score":5,"bucket":"other","tags":["多智能体博弈","纳什均衡","贝叶斯学习"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.18563","has_summary":true},{"id":"2603.20299","title":"HCAG: Hierarchical Abstraction and Retrieval-Augmented Generation on Theoretical Repositories with LLMs","zh_title":"HCAG：基于理论仓库的分层抽象与检索增强生成框架","primary_category":"cs.SE","date":"2026-03-19","score":0,"bucket":"other","tags":["代码生成","多智能体","检索增强生成"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.20299","has_summary":false},{"id":"2603.16142","title":"Parametric Social Identity Injection and Diversification in Public Opinion Simulation","zh_title":"参数化社会身份注入与多样化在舆论仿真中的应用","primary_category":"cs.CL","date":"2026-03-17","score":9,"bucket":"selected","tags":["LLM人类仿真","舆论模拟","社会身份注入"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2603.16142","has_summary":true},{"id":"2603.26701","title":"From Heard to Lived Opinions: Simulating Opinion Dynamics with Grounded LLM Agents in Economic Environments","zh_title":"从听闻到亲历：在经济环境中基于情境化LLM智能体模拟舆论动态","primary_category":"physics.soc-ph","date":"2026-03-17","score":7,"bucket":"pending","tags":["舆论动态","LLM智能体","社会模拟"],"rubric_hits":["A3","D3"],"abs_url":"https://arxiv.org/abs/2603.26701","has_summary":true},{"id":"2603.16659","title":"LLMs learn scientific taste from institutional traces across the social sciences","zh_title":"大语言模型从社会科学制度痕迹中学习科学品味","primary_category":"cs.AI","date":"2026-03-17","score":0,"bucket":"other","tags":["LLM评估","科学品味","制度痕迹"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2603.16659","has_summary":false},{"id":"2603.15852","title":"Playing Against the Machine: Cooperation, Communication, and Strategy Heterogeneity in Repeated Prisoner's Dilemma","zh_title":"与机器博弈：重复囚徒困境中的合作、沟通与策略异质性","primary_category":"econ.GN","date":"2026-03-16","score":0,"bucket":"other","tags":["人机互动","行为实验","囚徒困境"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.15852","has_summary":false},{"id":"2603.14903","title":"ExPosST: Explicit Positioning with Adaptive Masking for LLM-Based Simultaneous Machine Translation","zh_title":"ExPosST：基于显式定位与自适应掩码的大语言模型同声传译框架","primary_category":"cs.CL","date":"2026-03-16","score":0,"bucket":"other","tags":["同声传译","大语言模型","位置编码"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2603.14903","has_summary":false},{"id":"2603.13890","title":"Beyond Self-Interest: Modeling Social-Oriented Motivation for Human-like Multi-Agent Interactions","zh_title":"超越自利：建模社会导向动机以实现类人多智能体交互","primary_category":"cs.MA","date":"2026-03-14","score":5,"bucket":"other","tags":["社会模拟","多智能体","社会价值取向"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.13890","has_summary":true},{"id":"2603.12129","title":"Increasing intelligence in AI agents can worsen collective outcomes","zh_title":"AI智能体智能提升可能恶化集体结果","primary_category":"cs.AI","date":"2026-03-12","score":5,"bucket":"other","tags":["LLM智能体","集体行为","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.12129","has_summary":true},{"id":"2603.12000","title":"Credibility Matters: Motivations, Characteristics, and Influence Mechanisms of Crypto Key Opinion Leaders","zh_title":"可信度至关重要：加密货币关键意见领袖的动机、特征与影响机制","primary_category":"cs.SI","date":"2026-03-12","score":0,"bucket":"other","tags":["加密货币","意见领袖","主题分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.12000","has_summary":false},{"id":"2604.09609","title":"General-purpose LLMs as Models of Human Driver Behavior: The Case of Simplified Merging","zh_title":"通用大语言模型作为人类驾驶行为模型：简化合流场景案例","primary_category":"cs.AI","date":"2026-03-11","score":8,"bucket":"selected","tags":["LLM仿真","人类行为对照","驾驶行为建模"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.09609","has_summary":true},{"id":"2603.09884","title":"Benchmarking Political Persuasion Risks Across Frontier Large Language Models","zh_title":"跨前沿大语言模型的政治说服风险基准测试","primary_category":"cs.CL","date":"2026-03-10","score":9,"bucket":"selected","tags":["LLM仿真","政治说服","人类对照实验"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2603.09884","has_summary":true},{"id":"2603.09890","title":"Influencing LLM Multi-Agent Dialogue via Policy-Parameterized Prompts","zh_title":"通过策略参数化提示影响LLM多智能体对话","primary_category":"cs.AI","date":"2026-03-10","score":5,"bucket":"other","tags":["多智能体对话","社会模拟","提示参数化"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.09890","has_summary":true},{"id":"2603.08853","title":"LLM-Agent Interactions on Markets with Information Asymmetries","zh_title":"信息不对称市场中LLM智能体的互动研究","primary_category":"econ.GN","date":"2026-03-09","score":9,"bucket":"selected","tags":["LLM仿真","市场实验","人类数据对照"],"rubric_hits":["A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2603.08853","has_summary":true},{"id":"2604.15329","title":"Evaluating LLMs as Human Surrogates in Controlled Experiments","zh_title":"评估大语言模型作为受控实验中人类替代品的有效性","primary_category":"cs.HC","date":"2026-03-08","score":10,"bucket":"selected","tags":["LLM仿真","人类替代","实验对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2604.15329","has_summary":true},{"id":"2603.07444","title":"HLER: Human-in-the-Loop Economic Research via Multi-Agent Pipelines for Empirical Discovery","zh_title":"HLER：通过多智能体流水线进行人在回路的经济实证研究","primary_category":"cs.AI","date":"2026-03-08","score":5,"bucket":"other","tags":["多智能体系统","经济研究自动化","人在回路"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.07444","has_summary":true},{"id":"2604.22756","title":"Your Reviews Replicate You: LLM-Based Agents as Customer Digital Twins for Conjoint Analysis","zh_title":"你的评论复制你：基于LLM的客户数字孪生用于联合分析","primary_category":"cs.IR","date":"2026-03-06","score":9,"bucket":"selected","tags":["LLM仿真","消费者偏好","数字孪生"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2604.22756","has_summary":true},{"id":"2603.04276","title":"Causality Elicitation from Large Language Models","zh_title":"从大语言模型中提取因果关系","primary_category":"cs.LG","date":"2026-03-04","score":0,"bucket":"other","tags":["因果发现","LLM知识提取","文本挖掘"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2603.04276","has_summary":false},{"id":"2603.03623","title":"A Neural Topic Method Using a Large-Language-Model-in-the-Loop for Business Research","zh_title":"一种用于商业研究的循环大语言模型神经主题方法","primary_category":"cs.CL","date":"2026-03-04","score":0,"bucket":"other","tags":["主题建模","商业研究","文本分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2603.03623","has_summary":false},{"id":"2603.03585","title":"Belief-Sim: Towards Belief-Driven Simulation of Demographic Misinformation Susceptibility","zh_title":"Belief-Sim：面向信念驱动的人口统计错误信息易感性仿真","primary_category":"cs.CL","date":"2026-03-03","score":9,"bucket":"selected","tags":["LLM人类仿真","错误信息易感性","人口统计差异"],"rubric_hits":["A1","A2","B1","B3"],"abs_url":"https://arxiv.org/abs/2603.03585","has_summary":true},{"id":"2603.02711","title":"A Natural Language Agentic Approach to Study Affective Polarization","zh_title":"一种研究情感极化的自然语言智能体方法","primary_category":"cs.AI","date":"2026-03-03","score":5,"bucket":"other","tags":["情感极化","多智能体模拟","社交媒体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2603.02711","has_summary":true},{"id":"2603.03242","title":"Density-Guided Response Optimization: Community-Grounded Alignment via Implicit Acceptance Signals","zh_title":"密度引导的响应优化：基于隐式接受信号的社区对齐","primary_category":"cs.AI","date":"2026-03-03","score":0,"bucket":"other","tags":["语言模型对齐","社区规范","隐式偏好信号"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2603.03242","has_summary":false},{"id":"2603.02076","title":"When an AI Judges Your Work: The Hidden Costs of Algorithmic Assessment","zh_title":"当AI评判你的工作：算法评估的隐性代价","primary_category":"econ.GN","date":"2026-03-02","score":5,"bucket":"other","tags":["算法评估","人类行为","LLM作为评估者"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2603.02076","has_summary":false},{"id":"2603.21006","title":"How AI Systems Think About Education: Analyzing Latent Preference Patterns in Large Language Models","zh_title":"AI系统如何思考教育：分析大语言模型中的潜在偏好模式","primary_category":"cs.CY","date":"2026-02-28","score":5,"bucket":"other","tags":["LLM偏好测量","教育对齐","德尔菲法"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2603.21006","has_summary":true},{"id":"2602.21983","title":"Humanizing Robot Gaze Shifts: A Framework for Natural Gaze Shifts in Humanoid Robots","zh_title":"拟人化机器人注视转移：人形机器人自然注视转移框架","primary_category":"cs.RO","date":"2026-02-25","score":0,"bucket":"other","tags":["机器人注视","人机交互","运动生成"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2602.21983","has_summary":false},{"id":"2602.21091","title":"Can Interest-Bearing Positions Solve the Long-Horizon Problem in Prediction Markets?","zh_title":"计息头寸能否解决预测市场中的长期问题？","primary_category":"econ.GN","date":"2026-02-24","score":5,"bucket":"other","tags":["LLM代理模拟","预测市场","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2602.21091","has_summary":true},{"id":"2602.20440","title":"Intelligence Without Integrity: Why Capable LLMs May Undermine Reliability","zh_title":"有智无信：为何能力强的LLM可能损害可靠性","primary_category":"econ.GN","date":"2026-02-24","score":5,"bucket":"other","tags":["LLM可靠性","分析完整性","模型行为测量"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2602.20440","has_summary":true},{"id":"2602.19580","title":"Leap+Verify: Regime-Adaptive Speculative Weight Prediction for Accelerating Neural Network Training","zh_title":"Leap+Verify：用于加速神经网络训练的自适应权重预测框架","primary_category":"cs.LG","date":"2026-02-23","score":0,"bucket":"other","tags":["神经网络训练","投机执行","权重预测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2602.19580","has_summary":false},{"id":"2603.00113","title":"AI Agents Alone Are Not (Yet) Sufficient for Social Simulation","zh_title":"AI智能体单独尚不足以进行社会仿真","primary_category":"cs.MA","date":"2026-02-19","score":7,"bucket":"pending","tags":["社会仿真","LLM智能体","方法论批评"],"rubric_hits":["A3","D3"],"abs_url":"https://arxiv.org/abs/2603.00113","has_summary":true},{"id":"2602.15730","title":"Causal Effect Estimation with Latent Textual Treatments","zh_title":"基于潜在文本处理的因果效应估计","primary_category":"cs.CL","date":"2026-02-17","score":0,"bucket":"other","tags":["因果推断","文本处理","LLM生成"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.15730","has_summary":false},{"id":"2602.15312","title":"Extracting Consumer Insight from Text: A Large Language Model Approach to Emotion and Evaluation Measurement","zh_title":"从文本中提取消费者洞察：一种用于情感和评价测量的大语言模型方法","primary_category":"cs.CL","date":"2026-02-17","score":0,"bucket":"other","tags":["情感分析","消费者洞察","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.15312","has_summary":false},{"id":"2602.15173","title":"Mind the (DH) Gap! A Contrast in Risky Choices Between Reasoning and Conversational LLMs","zh_title":"注意(DH)差距！推理型与对话型LLM在风险选择上的对比","primary_category":"cs.AI","date":"2026-02-16","score":9,"bucket":"selected","tags":["LLM仿真","风险决策","人类对照实验"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2602.15173","has_summary":true},{"id":"2602.14043","title":"Beyond Static Snapshots: Dynamic Modeling and Forecasting of Group-Level Value Evolution with Large Language Models","zh_title":"超越静态快照：基于大语言模型的群体价值观动态建模与预测","primary_category":"cs.SI","date":"2026-02-15","score":9,"bucket":"selected","tags":["LLM人类仿真","价值观演化","社会模拟"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2602.14043","has_summary":true},{"id":"2602.13862","title":"Measuring Self-Rating Bias in LLM-Generated Survey Data: A Semantic Similarity Framework for Independent Scale Mapping","zh_title":"测量LLM生成调查数据中的自评偏差：一种独立量表映射的语义相似度框架","primary_category":"physics.soc-ph","date":"2026-02-14","score":8,"bucket":"selected","tags":["LLM仿真","调查数据","偏差评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.13862","has_summary":true},{"id":"2602.12490","title":"Transformer-based CoVaR: Systemic Risk in Textual Information","zh_title":"基于Transformer的CoVaR：文本信息中的系统性风险","primary_category":"econ.EM","date":"2026-02-13","score":0,"bucket":"other","tags":["金融风险预测","文本嵌入","Transformer模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.12490","has_summary":false},{"id":"2602.11939","title":"Do Large Language Models Adapt to Language Variation across Socioeconomic Status?","zh_title":"大语言模型能否适应社会经济地位带来的语言变异？","primary_category":"cs.CL","date":"2026-02-12","score":9,"bucket":"selected","tags":["LLM仿真","社会语言学","算法保真度"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.11939","has_summary":true},{"id":"2602.09362","title":"Behavioral Economics of AI: LLM Biases and Corrections","zh_title":"人工智能的行为经济学：大语言模型的偏差与校正","primary_category":"econ.GN","date":"2026-02-10","score":9,"bucket":"selected","tags":["LLM行为偏差","人类仿真","实验经济学"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2602.09362","has_summary":true},{"id":"2602.09802","title":"Would a Large Language Model Pay Extra for a View? Inferring Willingness to Pay from Subjective Choices","zh_title":"大语言模型会为景观多付钱吗？从主观选择推断支付意愿","primary_category":"cs.AI","date":"2026-02-10","score":9,"bucket":"selected","tags":["LLM仿真","支付意愿","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2602.09802","has_summary":true},{"id":"2602.07414","title":"Can LLMs Truly Embody Human Personality? Analyzing AI and Human Behavior Alignment in Dispute Resolution","zh_title":"LLM能真正体现人类人格吗？分析争议解决中AI与人类行为的一致性","primary_category":"cs.AI","date":"2026-02-07","score":9,"bucket":"selected","tags":["LLM人格仿真","人类行为对齐","争议解决"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.07414","has_summary":true},{"id":"2602.18462","title":"Assessing the Reliability of Persona-Conditioned LLMs as Synthetic Survey Respondents","zh_title":"评估基于人格条件的LLM作为合成调查受访者的可靠性","primary_category":"cs.CY","date":"2026-02-06","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","可靠性评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.18462","has_summary":true},{"id":"2602.18464","title":"How Well Can LLM Agents Simulate End-User Security and Privacy Attitudes and Behaviors?","zh_title":"LLM代理模拟终端用户安全与隐私态度及行为的效果如何？","primary_category":"cs.CY","date":"2026-02-06","score":9,"bucket":"selected","tags":["LLM仿真","人类行为对照","安全隐私"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.18464","has_summary":true},{"id":"2602.07238","title":"Is there \"Secret Sauce'' in Large Language Model Development?","zh_title":"大语言模型开发中是否存在“秘方”？","primary_category":"cs.AI","date":"2026-02-06","score":0,"bucket":"other","tags":["LLM性能分析","规模定律","算力与效率"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.07238","has_summary":false},{"id":"2602.13273","title":"MergePipe: A Budget-Aware Parameter Management System for Scalable LLM Merging","zh_title":"MergePipe：面向可扩展大语言模型合并的预算感知参数管理系统","primary_category":"cs.DB","date":"2026-02-05","score":0,"bucket":"other","tags":["模型合并","参数管理","系统优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2602.13273","has_summary":false},{"id":"2602.04674","title":"Overstating Attitudes, Ignoring Networks: LLM Biases in Simulating Misinformation Susceptibility","zh_title":"夸大态度，忽视网络：LLM在模拟错误信息易感性中的偏差","primary_category":"cs.SI","date":"2026-02-04","score":9,"bucket":"selected","tags":["LLM仿真","错误信息","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.04674","has_summary":true},{"id":"2602.03545","title":"Persona Generators: Generating Diverse Synthetic Personas for Arbitrary Contexts","zh_title":"人格生成器：为任意上下文生成多样化的合成人格","primary_category":"cs.AI","date":"2026-02-03","score":7,"bucket":"pending","tags":["合成人群生成","人类仿真","多样性覆盖"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2602.03545","has_summary":true},{"id":"2602.02977","title":"Aligning Forest and Trees in Images & Long Captions for Visually Grounded Understanding","zh_title":"对齐图像与长文本中的森林与树木以实现视觉基础理解","primary_category":"cs.CV","date":"2026-02-03","score":0,"bucket":"other","tags":["视觉-语言模型","细粒度对齐","图像文本检索"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.02977","has_summary":false},{"id":"2602.01684","title":"The Strategic Foresight of LLMs: Evidence from a Fully Prospective Venture Tournament","zh_title":"大语言模型的战略远见：来自全前瞻性创业锦标赛的证据","primary_category":"econ.GN","date":"2026-02-02","score":9,"bucket":"selected","tags":["LLM仿真","人类行为对照","创业预测"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2602.01684","has_summary":true},{"id":"2602.07023","title":"Behavioral Consistency Validation for LLM Agents: An Analysis of Trading-Style Switching through Stock-Market Simulation","zh_title":"LLM智能体行为一致性验证：基于股市模拟的交易风格切换分析","primary_category":"q-fin.TR","date":"2026-02-02","score":8,"bucket":"selected","tags":["LLM仿真","行为金融","智能体一致性"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2602.07023","has_summary":true},{"id":"2602.02606","title":"Gender Dynamics and Homophily in a Social Network of LLM Agents","zh_title":"LLM代理社交网络中的性别动态与同质性","primary_category":"cs.SI","date":"2026-02-02","score":7,"bucket":"pending","tags":["LLM代理","社会模拟","性别同质性"],"rubric_hits":["A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2602.02606","has_summary":true},{"id":"2602.02604","title":"AI Assisted Economics Measurement From Survey: Evidence from Public Employee Pension Choice","zh_title":"基于调查的人工智能辅助经济测量：来自公共雇员养老金选择的证据","primary_category":"econ.EM","date":"2026-02-02","score":0,"bucket":"other","tags":["LLM辅助测量","调查方法","经济构念"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.02604","has_summary":false},{"id":"2602.01797","title":"ORCH: many analyses, one merge-a deterministic multi-agent orchestrator for discrete-choice reasoning with EMA-guided routing","zh_title":"ORCH：一种用于离散选择推理的确定性多智能体协调框架","primary_category":"cs.AI","date":"2026-02-02","score":0,"bucket":"other","tags":["多智能体系统","推理增强","确定性路由"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2602.01797","has_summary":false},{"id":"2602.01022","title":"Calibrating Behavioral Parameters with Large Language Models","zh_title":"用大语言模型校准行为参数","primary_category":"econ.GN","date":"2026-02-01","score":9,"bucket":"selected","tags":["LLM仿真","行为经济学","人类基准对照"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2602.01022","has_summary":true},{"id":"2602.00948","title":"FinEvo: From Isolated Backtests to Ecological Market Games for Multi-Agent Financial Strategy Evolution","zh_title":"FinEvo：从孤立回测到生态市场博弈的多智能体金融策略演化","primary_category":"physics.soc-ph","date":"2026-02-01","score":5,"bucket":"other","tags":["多智能体模拟","金融市场演化","LLM agent"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2602.00948","has_summary":false},{"id":"2602.00685","title":"HumanStudy-Bench: Towards AI Agent Design for Participant Simulation","zh_title":"HumanStudy-Bench：面向参与者仿真的AI智能体设计基准","primary_category":"cs.AI","date":"2026-01-31","score":10,"bucket":"selected","tags":["LLM人类仿真","实验复现","基准测试"],"rubric_hits":["A1","A2","A3","A5","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2602.00685","has_summary":true},{"id":"2601.22812","title":"Stable Personas: Dual-Assessment of Temporal Stability in LLM-Based Human Simulation","zh_title":"稳定人格：基于LLM的人类仿真中时间稳定性的双重评估","primary_category":"cs.HC","date":"2026-01-30","score":8,"bucket":"selected","tags":["LLM仿真","人格稳定性","效度评估"],"rubric_hits":["A1","A2","B4"],"abs_url":"https://arxiv.org/abs/2601.22812","has_summary":true},{"id":"2601.23032","title":"Guided by Trajectories: Repairing and Rewarding Tool-Use Trajectories for Tool-Integrated Reasoning","zh_title":"轨迹引导：修复与奖励工具使用轨迹以实现工具集成推理","primary_category":"cs.AI","date":"2026-01-30","score":0,"bucket":"other","tags":["工具集成推理","轨迹优化","强化学习"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.23032","has_summary":false},{"id":"2601.21975","title":"Mind the Gap: How Elicitation Protocols Shape the Stated-Revealed Preference Gap in Language Models","zh_title":"注意差距：诱导协议如何塑造语言模型中陈述-显示偏好差距","primary_category":"cs.AI","date":"2026-01-29","score":5,"bucket":"other","tags":["语言模型偏好","诱导协议","陈述-显示偏好"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2601.21975","has_summary":true},{"id":"2601.20285","title":"Bank Runs With and Without Bank Failure","zh_title":"银行挤兑：有无银行倒闭的情形","primary_category":"econ.GN","date":"2026-01-28","score":0,"bucket":"other","tags":["银行挤兑","历史数据","文本分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.20285","has_summary":false},{"id":"2602.17676","title":"Epistemic Traps: Rational Misalignment Driven by Model Misspecification","zh_title":"认知陷阱：模型误设驱动的理性失配","primary_category":"cs.AI","date":"2026-01-27","score":4,"bucket":"other","tags":["AI对齐","多智能体系统","理论经济学"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2602.17676","has_summary":false},{"id":"2601.18027","title":"Sentipolis: Emotion-Aware Agents for Social Simulations","zh_title":"Sentipolis：用于社会模拟的情感感知智能体","primary_category":"cs.AI","date":"2026-01-25","score":5,"bucket":"other","tags":["社会模拟","情感建模","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2601.18027","has_summary":true},{"id":"2601.17527","title":"Bridging Expectation Signals: LLM-Based Experiments and a Behavioral Kalman Filter Framework","zh_title":"桥接预期信号：基于LLM的实验与行为卡尔曼滤波框架","primary_category":"econ.GN","date":"2026-01-24","score":8,"bucket":"selected","tags":["LLM经济代理","预期形成","行为偏差"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.17527","has_summary":true},{"id":"2604.06173","title":"Beyond Case Law: Evaluating Structure-Aware Retrieval and Safety in Statute-Centric Legal QA","zh_title":"超越判例法：评估法规中心的法律问答中的结构感知检索与安全性","primary_category":"cs.IR","date":"2026-01-24","score":0,"bucket":"other","tags":["法律问答","检索增强","模型安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2604.06173","has_summary":false},{"id":"2601.16355","title":"Identity, Cooperation and Framing Effects within Groups of Real and Simulated Humans","zh_title":"真实与模拟人类群体中的身份、合作与框架效应","primary_category":"cs.CL","date":"2026-01-22","score":9,"bucket":"selected","tags":["LLM人类仿真","社会困境博弈","行为实验复现"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2601.16355","has_summary":true},{"id":"2601.15793","title":"HumanLLM: Towards Personalized Understanding and Simulation of Human Nature","zh_title":"HumanLLM：迈向个性化理解与人性仿真","primary_category":"cs.CL","date":"2026-01-22","score":8,"bucket":"selected","tags":["人类行为仿真","个性化建模","社会模拟"],"rubric_hits":["A1","A2","B1"],"abs_url":"https://arxiv.org/abs/2601.15793","has_summary":true},{"id":"2601.15556","title":"LLM or Human? Perceptions of Trust and Information Quality in Research Summaries","zh_title":"LLM还是人类？研究摘要中信任与信息质量的感知","primary_category":"cs.CY","date":"2026-01-22","score":7,"bucket":"pending","tags":["LLM生成摘要","人类感知调查","信任与质量评估"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2601.15556","has_summary":true},{"id":"2601.15114","title":"From Who They Are to How They Act: Behavioral Traits in Generative Agent-Based Models of Social Media","zh_title":"从他们是谁到他们如何行动：基于生成式智能体的社交媒体模型中的行为特质","primary_category":"cs.MA","date":"2026-01-21","score":9,"bucket":"selected","tags":["LLM仿真","社交媒体模拟","行为特质"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2601.15114","has_summary":true},{"id":"2601.12727","title":"AI-exhibited Personality Traits Can Shape Human Self-concept through Conversations","zh_title":"AI展现的人格特质可通过对话塑造人类自我概念","primary_category":"cs.HC","date":"2026-01-19","score":8,"bucket":"selected","tags":["LLM仿真","人格影响","人机交互实验"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2601.12727","has_summary":true},{"id":"2601.12343","title":"How Well Do LLMs Predict Human Behavior? A Measure of their Pretrained Knowledge","zh_title":"LLM预测人类行为的效果如何？一种对其预训练知识的度量","primary_category":"econ.EM","date":"2026-01-18","score":9,"bucket":"selected","tags":["LLM仿真","人类行为预测","等效样本量"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2601.12343","has_summary":true},{"id":"2601.12339","title":"The Economics of Digital Intelligence Capital: Endogenous Depreciation and the Structural Jevons Paradox","zh_title":"数字智能资本的经济学：内生折旧与结构性杰文斯悖论","primary_category":"econ.GN","date":"2026-01-18","score":0,"bucket":"other","tags":["AI经济学","多智能体模拟","产业动态"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.12339","has_summary":false},{"id":"2601.12471","title":"Knowing When to Abstain: Medical LLMs Under Clinical Uncertainty","zh_title":"知止：临床不确定性下的医学大语言模型","primary_category":"cs.CL","date":"2026-01-18","score":0,"bucket":"other","tags":["医学问答","模型弃权","安全评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2601.12471","has_summary":false},{"id":"2601.11049","title":"Predicting Biased Human Decision-Making with Large Language Models in Conversational Settings","zh_title":"用大语言模型预测对话场景中的人类有偏决策","primary_category":"cs.HC","date":"2026-01-16","score":10,"bucket":"selected","tags":["LLM仿真","认知偏差","人类数据对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.11049","has_summary":true},{"id":"2601.15319","title":"Large Language Models as Simulative Agents for Neurodivergent Adult Psychometric Profiles","zh_title":"大语言模型作为神经多样性成人心理测量特征的仿真代理","primary_category":"q-bio.NC","date":"2026-01-16","score":9,"bucket":"selected","tags":["LLM仿真人类被试","心理测量","真实人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2601.15319","has_summary":true},{"id":"2601.15312","title":"Do people expect different behavior from large language models acting on their behalf? Evidence from norm elicitations in two canonical economic games","zh_title":"人们是否期望代表他们行事的大语言模型表现出不同行为？来自两个经典经济博弈中规范引出的证据","primary_category":"cs.GT","date":"2026-01-14","score":9,"bucket":"selected","tags":["LLM仿真","经济博弈","社会规范"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.15312","has_summary":true},{"id":"2601.09772","title":"Antisocial behavior towards large language model users: experimental evidence","zh_title":"针对大语言模型用户的反社会行为：实验证据","primary_category":"cs.AI","date":"2026-01-14","score":9,"bucket":"selected","tags":["LLM仿真","行为经济学","社会惩罚"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.09772","has_summary":true},{"id":"2601.09849","title":"Strategies of cooperation and defection in five large language models","zh_title":"五种大语言模型中的合作与背叛策略","primary_category":"cs.CY","date":"2026-01-14","score":9,"bucket":"selected","tags":["LLM仿真","行为博弈","人类数据对照"],"rubric_hits":["A1","A3","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2601.09849","has_summary":true},{"id":"2601.09119","title":"Contrastive Bi-Encoder Models for Multi-Label Skill Extraction: Enhancing ESCO Ontology Matching with BERT and Attention Mechanisms","zh_title":"用于多标签技能抽取的对比双编码器模型：利用BERT和注意力机制增强ESCO本体匹配","primary_category":"cs.CL","date":"2026-01-14","score":0,"bucket":"other","tags":["技能抽取","零样本分类","劳动力市场分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2601.09119","has_summary":false},{"id":"2602.02496","title":"The Hypocrisy Gap: Quantifying Divergence Between Internal Belief and Chain-of-Thought Explanation via Sparse Autoencoders","zh_title":"虚伪差距：通过稀疏自编码器量化内部信念与思维链解释之间的分歧","primary_category":"cs.CL","date":"2026-01-14","score":0,"bucket":"other","tags":["模型可解释性","忠实性检测","稀疏自编码器"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2602.02496","has_summary":false},{"id":"2601.07110","title":"The Need for a Socially-Grounded Persona Framework for User Simulation","zh_title":"面向用户仿真的社会根基化角色框架需求","primary_category":"cs.CL","date":"2026-01-12","score":9,"bucket":"selected","tags":["LLM人类仿真","社会心理角色","仿真偏差评估"],"rubric_hits":["A1","A2","A3","B1","B4"],"abs_url":"https://arxiv.org/abs/2601.07110","has_summary":true},{"id":"2601.07992","title":"Fake Date Tests: Can We Trust In-sample Accuracy of LLMs in Macroeconomic Forecasting?","zh_title":"虚假日期检验：我们能信任大语言模型在宏观经济预测中的样本内准确度吗？","primary_category":"econ.EM","date":"2026-01-12","score":0,"bucket":"other","tags":["LLM预测","宏观经济","样本内偏差"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2601.07992","has_summary":false},{"id":"2601.05050","title":"Large language models can effectively convince people to believe conspiracies","zh_title":"大语言模型能有效说服人们相信阴谋论","primary_category":"cs.AI","date":"2026-01-08","score":9,"bucket":"selected","tags":["LLM仿真","人类被试","说服实验"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.05050","has_summary":true},{"id":"2601.05104","title":"How Human is AI? Examining the Impact of Emotional Prompts on Artificial and Human and Responsiveness","zh_title":"AI有多人性化？情绪提示对人工与人类响应能力的影响研究","primary_category":"cs.CL","date":"2026-01-08","score":4,"bucket":"other","tags":["人机交互","情绪提示","ChatGPT行为"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2601.05104","has_summary":false},{"id":"2601.06180","title":"MixDPO: Modeling Preference Strength for Pluralistic Alignment","zh_title":"MixDPO：建模偏好强度以实现多元对齐","primary_category":"cs.LG","date":"2026-01-07","score":4,"bucket":"other","tags":["偏好对齐","强化学习","语言模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2601.06180","has_summary":false},{"id":"2601.03469","title":"Content vs. Form: What Drives the Writing Score Gap Across Socioeconomic Backgrounds? A Generated Panel Approach","zh_title":"内容与形式：什么驱动了社会经济背景下的写作分数差距？一种生成面板方法","primary_category":"econ.EM","date":"2026-01-06","score":5,"bucket":"other","tags":["LLM生成变体","教育测量","社会经济差距"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2601.03469","has_summary":true},{"id":"2601.02878","title":"Improving Financial Forecasting with a Synergistic LLM-Transformer Architecture: A Hybrid Approach to Stock Price Prediction","zh_title":"利用协同LLM-Transformer架构改进金融预测：一种混合股价预测方法","primary_category":"econ.TH","date":"2026-01-06","score":0,"bucket":"other","tags":["金融预测","LLM-Transformer混合模型","股价预测"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.02878","has_summary":false},{"id":"2601.01546","title":"Improving Behavioral Alignment in LLM Social Simulations via Context Formation and Navigation","zh_title":"通过情境形成与导航改进LLM社会仿真中的行为对齐","primary_category":"cs.AI","date":"2026-01-04","score":9,"bucket":"selected","tags":["LLM仿真","行为对齐","经济学实验"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2601.01546","has_summary":true},{"id":"2601.00240","title":"When Agents See Humans as the Outgroup: Belief-Dependent Bias in LLM-Powered Agents","zh_title":"当智能体将人类视为外群体：大语言模型智能体中基于信念的偏见","primary_category":"cs.AI","date":"2026-01-01","score":7,"bucket":"pending","tags":["LLM仿真","群际偏见","人机交互"],"rubric_hits":["A3","B4"],"abs_url":"https://arxiv.org/abs/2601.00240","has_summary":true},{"id":"2601.02407","title":"Evolving Personalities in Chaos: An LLM-Augmented Framework for Character Discovery in the Iterated Prisoners Dilemma under Environmental Stress","zh_title":"混沌中演化的人格：环境压力下迭代囚徒困境中基于大语言模型的角色发现框架","primary_category":"cs.NE","date":"2026-01-01","score":0,"bucket":"other","tags":["多智能体系统","演化博弈","角色分类"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.02407","has_summary":false},{"id":"2512.24856","title":"Advances in Agentic AI: Back to the Future","zh_title":"智能体人工智能的进展：回到未来","primary_category":"econ.TH","date":"2025-12-31","score":0,"bucket":"other","tags":["Agentic AI","多智能体系统","B2B转型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.24856","has_summary":false},{"id":"2512.23184","title":"From Model Choice to Model Belief: Establishing a New Measure for LLM-Based Research","zh_title":"从模型选择到模型信念：为基于LLM的研究建立新度量","primary_category":"cs.AI","date":"2025-12-29","score":8,"bucket":"selected","tags":["LLM仿真","需求估计","统计效率"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2512.23184","has_summary":true},{"id":"2512.23609","title":"Marriage Discourse on Chinese Social Media: An LLM-assisted Analysis","zh_title":"中国社交媒体上的婚姻话语：一项LLM辅助分析","primary_category":"econ.GN","date":"2025-12-29","score":0,"bucket":"other","tags":["内容分析","社交媒体","道德框架"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2512.23609","has_summary":false},{"id":"2512.22725","title":"Mitigating Social Desirability Bias in Random Silicon Sampling","zh_title":"缓解随机硅采样中的社会赞许性偏差","primary_category":"cs.CL","date":"2025-12-27","score":10,"bucket":"selected","tags":["硅采样","社会赞许性偏差","人类数据对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2512.22725","has_summary":true},{"id":"2512.21316","title":"Scaling Laws for Economic Productivity: Experimental Evidence in LLM-Assisted Consulting, Data Analyst, and Management Tasks","zh_title":"经济生产力的规模法则：LLM辅助咨询、数据分析和管理任务的实验证据","primary_category":"econ.GN","date":"2025-12-24","score":0,"bucket":"other","tags":["LLM辅助生产力","规模法则","人机协作"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.21316","has_summary":true},{"id":"2512.21402","title":"Understanding Virality: A Rubric based Vision-Language Model Framework for Short-Form Edutainment Evaluation","zh_title":"理解病毒式传播：基于量规的视觉语言模型框架用于短视频寓教于乐评估","primary_category":"cs.CV","date":"2025-12-24","score":0,"bucket":"other","tags":["视频理解","视觉语言模型","内容评估"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2512.21402","has_summary":false},{"id":"2512.19937","title":"Interpolative Decoding: Exploring the Spectrum of Personality Traits in LLMs","zh_title":"插值解码：探索大语言模型中人格特质的谱系","primary_category":"cs.AI","date":"2025-12-23","score":9,"bucket":"selected","tags":["LLM仿真","人格特质","经济博弈"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2512.19937","has_summary":true},{"id":"2512.19675","title":"Multimodal LLMs for Historical Dataset Construction from Archival Image Scans: German Patents (1877-1918)","zh_title":"利用多模态大语言模型从档案图像扫描件构建历史数据集：德国专利（1877-1918）","primary_category":"econ.GN","date":"2025-12-22","score":0,"bucket":"other","tags":["历史数据构建","多模态LLM","自动化数据提取"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.19675","has_summary":false},{"id":"2512.19484","title":"Structured Event Representation and Stock Return Predictability","zh_title":"结构化事件表示与股票收益可预测性","primary_category":"econ.GN","date":"2025-12-22","score":0,"bucket":"other","tags":["LLM金融应用","股票收益预测","事件表示"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2512.19484","has_summary":false},{"id":"2601.00810","title":"Can Large Language Models Improve Venture Capital Exit Timing After IPO?","zh_title":"大型语言模型能否改善风险投资IPO后的退出时机？","primary_category":"q-fin.PM","date":"2025-12-22","score":0,"bucket":"other","tags":["LLM应用","风险投资","退出决策"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2601.00810","has_summary":true},{"id":"2512.14306","title":"Inflation Attitudes of Large Language Models","zh_title":"大语言模型的通胀态度","primary_category":"cs.CL","date":"2025-12-16","score":10,"bucket":"selected","tags":["LLM仿真","通胀预期","人类数据对照"],"rubric_hits":["A1","A2","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2512.14306","has_summary":true},{"id":"2512.14562","title":"Polypersona: Persona-Grounded LLM for Synthetic Survey Responses","zh_title":"Polypersona：基于人格的LLM合成调查回复生成框架","primary_category":"cs.CL","date":"2025-12-16","score":7,"bucket":"pending","tags":["合成调查数据","人格条件生成","人类仿真"],"rubric_hits":["A1","B1"],"abs_url":"https://arxiv.org/abs/2512.14562","has_summary":true},{"id":"2512.12597","title":"AgentSHAP: Interpreting LLM Agent Tool Importance with Monte Carlo Shapley Value Estimation","zh_title":"AgentSHAP：用蒙特卡洛Shapley值估计解释LLM智能体工具重要性","primary_category":"cs.AI","date":"2025-12-14","score":0,"bucket":"other","tags":["可解释性","多智能体","工具调用"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.12597","has_summary":false},{"id":"2512.12444","title":"Can GPT replace human raters? Validity and reliability of machine-generated norms for metaphors","zh_title":"GPT能否替代人类评分员？机器生成隐喻常模的效度与信度","primary_category":"cs.CL","date":"2025-12-13","score":5,"bucket":"other","tags":["LLM标注","隐喻评分","效度信度"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2512.12444","has_summary":true},{"id":"2512.11943","title":"How AI Agents Follow the Herd of AI? Network Effects, History, and Machine Optimism","zh_title":"AI智能体如何跟随AI的羊群效应？网络效应、历史与机器乐观主义","primary_category":"cs.MA","date":"2025-12-12","score":5,"bucket":"other","tags":["LLM智能体","网络效应博弈","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2512.11943","has_summary":true},{"id":"2512.09652","title":"Measuring Corruption from Text Data","zh_title":"从文本数据中测量腐败","primary_category":"econ.GN","date":"2025-12-10","score":0,"bucket":"other","tags":["腐败测量","文本分析","审计报告"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2512.09652","has_summary":false},{"id":"2512.08345","title":"The High Cost of Incivility: Quantifying Interaction Inefficiency via Multi-Agent Monte Carlo Simulations","zh_title":"不文明行为的高昂代价：通过多智能体蒙特卡洛模拟量化互动低效","primary_category":"cs.AI","date":"2025-12-09","score":5,"bucket":"other","tags":["多智能体模拟","社会摩擦","LLM仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2512.08345","has_summary":true},{"id":"2512.05659","title":"Beyond Automation: Redesigning Jobs with LLMs to Enhance Productivity","zh_title":"超越自动化：利用大语言模型重新设计工作以提升生产力","primary_category":"econ.GN","date":"2025-12-05","score":0,"bucket":"other","tags":["AI暴露度","工作重新设计","生产力"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.05659","has_summary":false},{"id":"2512.04988","title":"When AI Agents Compete for Jobs: Strategic Capabilities and Economic Dynamics of AI Labour Markets","zh_title":"当AI智能体竞争工作：AI劳动力市场的战略能力与经济动态","primary_category":"cs.MA","date":"2025-12-04","score":5,"bucket":"other","tags":["AI劳动力市场","多智能体模拟","经济仿真"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2512.04988","has_summary":true},{"id":"2512.06033","title":"Sell Data to AI Algorithms Without Revealing It: Secure Data Valuation and Sharing via Homomorphic Encryption","zh_title":"在不泄露数据的情况下向AI算法出售数据：基于同态加密的安全数据估值与共享","primary_category":"cs.CR","date":"2025-12-04","score":0,"bucket":"other","tags":["隐私保护","数据估值","同态加密"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.06033","has_summary":false},{"id":"2512.03568","title":"Synthetic Cognitive Walkthrough: Aligning Large Language Model Performance with Human Cognitive Walkthrough","zh_title":"合成认知走查：将大语言模型表现与人类认知走查对齐","primary_category":"cs.HC","date":"2025-12-03","score":7,"bucket":"pending","tags":["LLM仿真","认知走查","人机对照"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2512.03568","has_summary":true},{"id":"2512.04142","title":"From FLOPs to Footprints: The Resource Cost of Artificial Intelligence","zh_title":"从FLOPs到足迹：人工智能的资源成本","primary_category":"cs.CY","date":"2025-12-03","score":0,"bucket":"other","tags":["AI硬件","材料足迹","环境影响"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2512.04142","has_summary":false},{"id":"2512.11827","title":"Assessing Greenspace Attractiveness with ChatGPT, Claude, and Gemini: Do AI Models Reflect Human Perceptions?","zh_title":"用ChatGPT、Claude和Gemini评估绿地吸引力：AI模型是否反映人类感知？","primary_category":"cs.CY","date":"2025-12-02","score":9,"bucket":"selected","tags":["LLM人类仿真","绿地感知评估","AI与人类对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2512.11827","has_summary":true},{"id":"2512.07890","title":"CrowdLLM: Building LLM-Based Digital Populations Augmented with Generative Models","zh_title":"CrowdLLM：结合生成模型构建基于LLM的数字人群","primary_category":"cs.MA","date":"2025-12-02","score":9,"bucket":"selected","tags":["LLM仿真","数字人群","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2512.07890","has_summary":true},{"id":"2601.11542","title":"The Credibility Revolution in Political Science","zh_title":"政治学中的可信性革命","primary_category":"cs.DL","date":"2025-12-02","score":0,"bucket":"other","tags":["文献计量","研究设计","NLP应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2601.11542","has_summary":false},{"id":"2512.01107","title":"Foundation Priors","zh_title":"基础先验：将基础模型输出作为结构化主观先验","primary_category":"cs.AI","date":"2025-11-30","score":5,"bucket":"other","tags":["合成数据","贝叶斯先验","方法论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2512.01107","has_summary":true},{"id":"2511.21218","title":"Can Finetuing LLMs on Small Human Samples Increase Heterogeneity, Alignment, and Belief-Action Coherence?","zh_title":"在小规模人类样本上微调LLM能否增加异质性、对齐度和信念-行动一致性？","primary_category":"cs.CL","date":"2025-11-26","score":9,"bucket":"selected","tags":["LLM人类仿真","行为实验","微调偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2511.21218","has_summary":true},{"id":"2511.16278","title":"\"To Survive, I Must Defect\": Jailbreaking LLMs via the Game-Theory Scenarios","zh_title":"“为了生存，我必须背叛”：通过博弈论场景越狱大语言模型","primary_category":"cs.CR","date":"2025-11-20","score":0,"bucket":"other","tags":["越狱攻击","博弈论","大语言模型安全"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2511.16278","has_summary":false},{"id":"2511.14359","title":"Towards LLM-Based Usability Analysis for Recommender User Interfaces","zh_title":"面向推荐系统用户界面的基于大语言模型的可用性分析","primary_category":"cs.HC","date":"2025-11-18","score":0,"bucket":"other","tags":["可用性评估","多模态LLM","推荐系统"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2511.14359","has_summary":false},{"id":"2511.09381","title":"Self-Correcting Large Language Models: Generation vs. Multiple Choice","zh_title":"自纠正大语言模型：生成与多项选择的对比","primary_category":"cs.CL","date":"2025-11-12","score":0,"bucket":"other","tags":["自纠正","NLP评测","推理任务"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2511.09381","has_summary":false},{"id":"2511.09047","title":"Preference is More Than Comparisons: Rethinking Dueling Bandits with Augmented Human Feedback","zh_title":"偏好不止于比较：重新思考基于增强人类反馈的对决赌博机","primary_category":"cs.LG","date":"2025-11-12","score":0,"bucket":"other","tags":["对决赌博机","交互式偏好获取","反馈增强"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2511.09047","has_summary":false},{"id":"2511.08785","title":"Making Talk Cheap: Generative AI and Labor Market Signaling","zh_title":"让谈话变得廉价：生成式AI与劳动力市场信号传递","primary_category":"econ.GN","date":"2025-11-11","score":5,"bucket":"other","tags":["劳动力市场","信号传递","结构模型"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2511.08785","has_summary":true},{"id":"2511.06260","title":"LLM-Guided Reinforcement Learning with Representative Agents for Traffic Modeling","zh_title":"基于代表性智能体的LLM引导强化学习交通建模","primary_category":"cs.GT","date":"2025-11-09","score":7,"bucket":"pending","tags":["LLM仿真","交通行为建模","多智能体系统"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2511.06260","has_summary":true},{"id":"2511.05766","title":"Anchors in the Machine: Behavioral and Attributional Evidence of Anchoring Bias in LLMs","zh_title":"机器中的锚定：LLM中锚定偏差的行为与归因证据","primary_category":"cs.AI","date":"2025-11-07","score":8,"bucket":"selected","tags":["锚定偏差","LLM行为仿真","可解释性"],"rubric_hits":["A1","A2","B4"],"abs_url":"https://arxiv.org/abs/2511.05766","has_summary":true},{"id":"2511.03758","title":"Leveraging LLM-based agents for social science research: insights from citation network simulations","zh_title":"利用基于大语言模型的智能体进行社会科学研究：来自引文网络模拟的见解","primary_category":"physics.soc-ph","date":"2025-11-05","score":9,"bucket":"selected","tags":["LLM仿真","引文网络","社会科学实验"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2511.03758","has_summary":true},{"id":"2511.02458","title":"Prompting for Policy: Forecasting Macroeconomic Scenarios with Synthetic LLM Personas","zh_title":"用合成LLM角色预测宏观经济情景的政策提示","primary_category":"cs.CL","date":"2025-11-04","score":9,"bucket":"selected","tags":["LLM仿真","宏观经济预测","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2511.02458","has_summary":true},{"id":"2510.26727","title":"Neither Consent nor Property: A Policy Lab for Data Law","zh_title":"非同意非财产：数据法律的政策实验室","primary_category":"econ.GN","date":"2025-10-30","score":5,"bucket":"other","tags":["LLM仿真","政策实验","数据市场"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2510.26727","has_summary":true},{"id":"2512.08939","title":"Assessing the Human-Likeness of LLM-Driven Digital Twins in Simulating Health Care System Trust","zh_title":"评估LLM驱动的数字孪生在模拟医疗系统信任中的人类相似性","primary_category":"cs.HC","date":"2025-10-27","score":9,"bucket":"selected","tags":["LLM仿真","人类数字孪生","医疗系统信任"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2512.08939","has_summary":true},{"id":"2510.18155","title":"LLM-Based Multi-Agent System for Simulating and Analyzing Marketing and Consumer Behavior","zh_title":"基于大语言模型的多智能体系统用于模拟和分析营销与消费者行为","primary_category":"cs.AI","date":"2025-10-20","score":5,"bucket":"other","tags":["LLM仿真","消费者行为","多智能体"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2510.18155","has_summary":true},{"id":"2510.16551","title":"From Reviews to Actionable Insights: An LLM-Based Approach for Attribute and Feature Extraction","zh_title":"从评论到可操作洞察：基于大语言模型的属性与特征提取方法","primary_category":"stat.ML","date":"2025-10-18","score":5,"bucket":"other","tags":["LLM标注","文本挖掘","营销分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2510.16551","has_summary":true},{"id":"2510.14301","title":"A Guardrail for Safety Preservation: When Safety-Sensitive Subspace Meets Harmful-Resistant Null-Space","zh_title":"安全保护的护栏：当安全敏感子空间遇到有害抵抗零空间","primary_category":"cs.AI","date":"2025-10-16","score":0,"bucket":"other","tags":["安全对齐","微调","模型安全"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2510.14301","has_summary":false},{"id":"2510.13091","title":"Unmasking Hiring Bias: Platform Data Analysis and Controlled Experiments on Bias in Online Freelance Marketplaces via RAG-LLM Generated Contents","zh_title":"揭示招聘偏见：基于RAG-LLM生成内容的在线自由职业市场偏见平台数据分析与受控实验","primary_category":"cs.HC","date":"2025-10-15","score":5,"bucket":"other","tags":["LLM生成合成数据","招聘偏见","受控实验"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2510.13091","has_summary":true},{"id":"2510.13011","title":"Deliberate Lab: A Platform for Real-Time Human-AI Social Experiments","zh_title":"Deliberate Lab：一个实时人类-AI社会实验平台","primary_category":"cs.HC","date":"2025-10-14","score":7,"bucket":"pending","tags":["人机混合实验","集体决策","LLM代理"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2510.13011","has_summary":true},{"id":"2510.12189","title":"Agent-Based Simulation of a Financial Market with Large Language Models","zh_title":"基于大语言模型的金融市场智能体仿真","primary_category":"cs.CE","date":"2025-10-14","score":7,"bucket":"pending","tags":["LLM仿真","行为金融","智能体市场"],"rubric_hits":["A3","B2"],"abs_url":"https://arxiv.org/abs/2510.12189","has_summary":true},{"id":"2510.08338","title":"LLMs Reproduce Human Purchase Intent via Semantic Similarity Elicitation of Likert Ratings","zh_title":"大语言模型通过语义相似度引出李克特评分复现人类购买意向","primary_category":"cs.AI","date":"2025-10-09","score":9,"bucket":"selected","tags":["LLM仿真","消费者调查","人类数据对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2510.08338","has_summary":true},{"id":"2510.08236","title":"The Hidden Bias: A Study on Explicit and Implicit Political Stereotypes in Large Language Models","zh_title":"隐藏的偏见：大语言模型中显性与隐性政治刻板印象研究","primary_category":"cs.LG","date":"2025-10-09","score":5,"bucket":"other","tags":["政治偏见","刻板印象","LLM评估"],"rubric_hits":["D2"],"abs_url":"https://arxiv.org/abs/2510.08236","has_summary":true},{"id":"2510.08191","title":"Training-Free Group Relative Policy Optimization","zh_title":"免训练的分组相对策略优化","primary_category":"cs.CL","date":"2025-10-09","score":0,"bucket":"other","tags":["多智能体强化学习","工具调用","免训练优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2510.08191","has_summary":false},{"id":"2510.07733","title":"SurveyG: A Multi-Agent LLM Framework with Hierarchical Citation Graph for Automated Survey Generation","zh_title":"SurveyG：基于层次引用图的多智能体LLM框架用于自动生成综述","primary_category":"cs.AI","date":"2025-10-09","score":0,"bucket":"other","tags":["自动综述生成","多智能体系统","引用图"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2510.07733","has_summary":false},{"id":"2510.06903","title":"When Machines Meet Each Other: Network Effects and the Strategic Role of History in Multi-Agent AI","zh_title":"当机器相遇：多智能体AI中的网络效应与历史的战略角色","primary_category":"econ.GN","date":"2025-10-08","score":7,"bucket":"pending","tags":["LLM仿真","网络效应博弈","多智能体"],"rubric_hits":["A3","B2"],"abs_url":"https://arxiv.org/abs/2510.06903","has_summary":true},{"id":"2510.06151","title":"LLMs as Policy-Agnostic Teammates: A Case Study in Human Proxy Design for Heterogeneous Agent Teams","zh_title":"作为策略无关队友的大语言模型：异构智能体团队中人类代理设计的案例研究","primary_category":"cs.LG","date":"2025-10-07","score":8,"bucket":"selected","tags":["LLM人类仿真","行为博弈","人机协作"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2510.06151","has_summary":true},{"id":"2510.03231","title":"Reward Models are Metrics in a Trench Coat","zh_title":"奖励模型是披着风衣的度量指标","primary_category":"cs.CL","date":"2025-10-03","score":0,"bucket":"other","tags":["奖励模型","评估指标","元评估"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2510.03231","has_summary":false},{"id":"2510.01115","title":"Exploring Network-Knowledge Graph Duality: A Case Study in Agentic Supply Chain Risk Analysis","zh_title":"探索网络-知识图谱对偶性：以智能体供应链风险分析为例","primary_category":"cs.AI","date":"2025-10-01","score":0,"bucket":"other","tags":["多智能体系统","供应链风险","知识图谱"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2510.01115","has_summary":false},{"id":"2509.25709","title":"Leveraging LLMs to Improve Experimental Design: A Generative Stratification Approach","zh_title":"利用大语言模型改进实验设计：一种生成式分层方法","primary_category":"econ.EM","date":"2025-09-30","score":5,"bucket":"other","tags":["实验设计","分层抽样","合成数据"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2509.25709","has_summary":true},{"id":"2510.02343","title":"$\\texttt{BluePrint}$: A Social Media User Dataset for LLM Persona Evaluation and Training","zh_title":"BluePrint：用于LLM角色评估与训练的社交媒体用户数据集","primary_category":"cs.CL","date":"2025-09-27","score":8,"bucket":"selected","tags":["LLM仿真","社交媒体模拟","行为保真度"],"rubric_hits":["A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2510.02343","has_summary":true},{"id":"2510.02331","title":"Synthetic Dialogue Generation for Interactive Conversational Elicitation & Recommendation (ICER)","zh_title":"面向交互式对话引导与推荐的合成对话生成","primary_category":"cs.CL","date":"2025-09-26","score":4,"bucket":"other","tags":["对话生成","推荐系统","用户模拟器"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2510.02331","has_summary":false},{"id":"2509.13712","title":"Inject, Fork, Compare: Defining an Interaction Vocabulary for Multi-Agent Simulation Platforms","zh_title":"注入、分叉、比较：为多智能体仿真平台定义交互词汇","primary_category":"cs.MA","date":"2025-09-17","score":5,"bucket":"other","tags":["多智能体仿真","社会模拟","交互操作"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2509.13712","has_summary":true},{"id":"2509.13397","title":"The threat of analytic flexibility in using large language models to simulate human data","zh_title":"使用大语言模型模拟人类数据时分析灵活性的威胁","primary_category":"cs.CY","date":"2025-09-16","score":9,"bucket":"selected","tags":["硅样本","分析灵活性","仿真保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2509.13397","has_summary":true},{"id":"2509.12350","title":"Knowledge Graph Tokenization for Behavior-Aware Generative Next POI Recommendation","zh_title":"面向行为感知的生成式下一兴趣点推荐的知识图谱分词","primary_category":"cs.IR","date":"2025-09-15","score":0,"bucket":"other","tags":["POI推荐","知识图谱","LLM微调"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2509.12350","has_summary":false},{"id":"2509.11311","title":"Prompts to Proxies: Emulating Human Preferences via a Compact LLM Ensemble","zh_title":"从提示到代理：通过紧凑LLM集成模拟人类偏好","primary_category":"cs.AI","date":"2025-09-14","score":9,"bucket":"selected","tags":["LLM人类仿真","偏好重建","社会调查"],"rubric_hits":["A1","A2","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2509.11311","has_summary":true},{"id":"2509.09871","title":"Emulating Public Opinion: A Proof-of-Concept of AI-Generated Synthetic Survey Responses for the Chilean Case","zh_title":"模拟民意：智利案例中AI生成合成调查回答的概念验证","primary_category":"cs.CL","date":"2025-09-11","score":10,"bucket":"selected","tags":["LLM仿真","调查方法","算法保真度"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2509.09871","has_summary":true},{"id":"2509.06337","title":"Large Language Models as Virtual Survey Respondents: Evaluating Sociodemographic Response Generation","zh_title":"大语言模型作为虚拟调查受访者：评估社会人口响应生成","primary_category":"cs.AI","date":"2025-09-08","score":9,"bucket":"selected","tags":["LLM仿真","调查方法","社会人口模拟"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2509.06337","has_summary":true},{"id":"2509.03736","title":"Are LLM Agents Behaviorally Coherent? Latent Profiles for Social Simulation","zh_title":"LLM代理行为一致吗？社会模拟的潜在画像","primary_category":"cs.AI","date":"2025-09-03","score":9,"bucket":"selected","tags":["LLM仿真","行为一致性","人类被试替代"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2509.03736","has_summary":true},{"id":"2509.02879","title":"Artificial or Human Intelligence?","zh_title":"人工智能还是人类智能？","primary_category":"econ.TH","date":"2025-09-02","score":0,"bucket":"other","tags":["AI教育","学生激励","多智能体"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2509.02879","has_summary":false},{"id":"2509.01813","title":"ShortageSim: Simulating Drug Shortages under Information Asymmetry","zh_title":"ShortageSim：信息不对称下的药品短缺仿真","primary_category":"cs.MA","date":"2025-09-01","score":7,"bucket":"pending","tags":["LLM仿真","供应链博弈","政策评估"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2509.01813","has_summary":true},{"id":"2510.07321","title":"How human is the machine? Evidence from 66,000 Conversations with Large Language Models","zh_title":"机器有多像人？来自66000次与大语言模型对话的证据","primary_category":"cs.HC","date":"2025-08-31","score":9,"bucket":"selected","tags":["LLM仿真","认知偏差","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2510.07321","has_summary":true},{"id":"2510.06222","title":"Inducing State Anxiety in LLM Agents Reproduces Human-Like Biases in Consumer Decision-Making","zh_title":"在LLM智能体中诱导状态焦虑可复现消费者决策中的人类偏差","primary_category":"cs.HC","date":"2025-08-30","score":9,"bucket":"selected","tags":["LLM仿真","消费者行为","焦虑偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2510.06222","has_summary":true},{"id":"2509.00462","title":"AI Self-preferencing in Algorithmic Hiring: Empirical Evidence and Insights","zh_title":"算法招聘中的人工智能自我偏好：实证证据与见解","primary_category":"cs.CY","date":"2025-08-30","score":7,"bucket":"pending","tags":["LLM仿真","招聘实验","算法偏差"],"rubric_hits":["A1","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2509.00462","has_summary":true},{"id":"2509.02605","title":"Synthetic Founders: AI-Generated Social Simulations for Startup Validation Research in Computational Social Science","zh_title":"合成创始人：用于计算社会科学中创业验证研究的AI生成社会仿真","primary_category":"cs.MA","date":"2025-08-29","score":9,"bucket":"selected","tags":["LLM人类仿真","创业者访谈对照","仿真保真度评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2509.02605","has_summary":true},{"id":"2509.02596","title":"Introducing LCOAI: A Standardized Economic Metric for Evaluating AI Deployment Costs","zh_title":"引入LCOAI：评估AI部署成本的标准化经济指标","primary_category":"econ.GN","date":"2025-08-29","score":0,"bucket":"other","tags":["AI经济学","成本指标","部署评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2509.02596","has_summary":false},{"id":"2508.20234","title":"Validating Generative Agent-Based Models for Logistics and Supply Chain Management Research","zh_title":"验证基于生成式智能体的物流与供应链管理研究模型","primary_category":"cs.MA","date":"2025-08-27","score":10,"bucket":"selected","tags":["LLM人类仿真","等效性验证","供应链管理"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2508.20234","has_summary":true},{"id":"2508.19004","title":"AI Models Exceed Individual Human Accuracy in Predicting Everyday Social Norms","zh_title":"AI模型在预测日常社会规范方面超越个体人类准确性","primary_category":"cs.AI","date":"2025-08-26","score":8,"bucket":"selected","tags":["社会规范预测","人类数据对照","算法偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2508.19004","has_summary":true},{"id":"2509.00074","title":"Language and Experience: A Computational Model of Social Learning in Complex Tasks","zh_title":"语言与经验：复杂任务中社会学习的计算模型","primary_category":"cs.AI","date":"2025-08-26","score":7,"bucket":"pending","tags":["LLM仿真","人类行为对照","社会学习"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2509.00074","has_summary":true},{"id":"2508.17322","title":"Chinese Court Simulation with LLM-Based Agent System","zh_title":"基于大语言模型智能体的中国法庭仿真","primary_category":"cs.CY","date":"2025-08-24","score":5,"bucket":"other","tags":["法庭仿真","多智能体","法律预测"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2508.17322","has_summary":true},{"id":"2508.16172","title":"Graph RAG as Human Choice Model: Building a Data-Driven Mobility Agent with Preference Chain","zh_title":"图RAG作为人类选择模型：构建数据驱动的出行智能体与偏好链","primary_category":"cs.AI","date":"2025-08-22","score":8,"bucket":"selected","tags":["LLM仿真","出行行为","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2508.16172","has_summary":true},{"id":"2508.15926","title":"Noise, Adaptation, and Strategy: Assessing LLM Fidelity in Decision-Making","zh_title":"噪声、适应与策略：评估LLM在决策中的保真度","primary_category":"cs.CE","date":"2025-08-21","score":9,"bucket":"selected","tags":["LLM仿真","行为保真度","经济学实验"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2508.15926","has_summary":true},{"id":"2508.12045","title":"Large Language Models Enable Design of Personalized Nudges across Cultures","zh_title":"大语言模型助力跨文化个性化助推设计","primary_category":"cs.CY","date":"2025-08-16","score":9,"bucket":"selected","tags":["LLM仿真","行为助推","跨文化实验"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2508.12045","has_summary":true},{"id":"2508.11873","title":"SimInterview: Transforming Business Education through Large Language Model-Based Simulated Multilingual Interview Training System","zh_title":"SimInterview：基于大语言模型的模拟多语种面试训练系统助力商业教育转型","primary_category":"cs.CY","date":"2025-08-16","score":0,"bucket":"other","tags":["面试训练","角色扮演","教育技术"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2508.11873","has_summary":false},{"id":"2508.09713","title":"Evaluating the Role of Large Language Models in Legal Practice in India","zh_title":"评估大语言模型在印度法律实践中的作用","primary_category":"cs.CL","date":"2025-08-13","score":5,"bucket":"other","tags":["LLM法律任务","人类对照","替代劳动"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2508.09713","has_summary":false},{"id":"2508.08486","title":"Beyond Ordinal Preferences: Why Alignment Needs Cardinal Human Feedback","zh_title":"超越序数偏好：为何对齐需要基数人类反馈","primary_category":"cs.AI","date":"2025-08-11","score":0,"bucket":"other","tags":["LLM对齐","偏好学习","基数反馈"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2508.08486","has_summary":false},{"id":"2508.06635","title":"Valid Inference with Imperfect Synthetic Data","zh_title":"不完美合成数据的有效推断","primary_category":"cs.LG","date":"2025-08-08","score":8,"bucket":"selected","tags":["合成数据","统计推断","计算社会科学"],"rubric_hits":["A1","A5","B1","B2"],"abs_url":"https://arxiv.org/abs/2508.06635","has_summary":true},{"id":"2508.02766","title":"The Generative Reasonable Person","zh_title":"生成式理性人","primary_category":"cs.CY","date":"2025-08-04","score":10,"bucket":"selected","tags":["LLM仿真","人类被试替代","法律判断"],"rubric_hits":["A1","A2","A5","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2508.02766","has_summary":true},{"id":"2508.05670","title":"Can LLMs effectively provide game-theoretic-based scenarios for cybersecurity?","zh_title":"大语言模型能否有效提供基于博弈论的网络安全场景？","primary_category":"cs.CR","date":"2025-08-04","score":5,"bucket":"other","tags":["LLM博弈行为","网络安全","社会模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2508.05670","has_summary":true},{"id":"2508.00485","title":"A Frame for Communication Control","zh_title":"一种通信控制框架","primary_category":"cs.CY","date":"2025-08-01","score":0,"bucket":"other","tags":["LLM监管","通信框架","法律术语"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2508.00485","has_summary":false},{"id":"2507.22049","title":"Validating Generative Agent-Based Models of Social Norm Enforcement: From Replication to Novel Predictions","zh_title":"验证基于生成式智能体的社会规范执行模型：从复现到新预测","primary_category":"cs.MA","date":"2025-07-29","score":9,"bucket":"selected","tags":["LLM仿真","社会规范","行为博弈"],"rubric_hits":["A1","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2507.22049","has_summary":true},{"id":"2507.21432","title":"Towards Locally Deployable Fine-Tuned Causal Large Language Models for Mode Choice Behaviour","zh_title":"面向出行方式选择行为的本地可部署微调因果大语言模型研究","primary_category":"cs.CL","date":"2025-07-29","score":7,"bucket":"pending","tags":["LLM行为预测","交通方式选择","人类数据对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2507.21432","has_summary":true},{"id":"2507.17564","title":"Decoding Consumer Preferences Using Attention-Based Language Models","zh_title":"使用基于注意力的语言模型解码消费者偏好","primary_category":"econ.EM","date":"2025-07-23","score":0,"bucket":"other","tags":["需求估计","语言模型","结构模型"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2507.17564","has_summary":false},{"id":"2507.17024","title":"Write, Rank, or Rate: Comparing Methods for Studying Visualization Affordances","zh_title":"写、排或评：比较研究可视化可供性的方法","primary_category":"cs.HC","date":"2025-07-22","score":5,"bucket":"other","tags":["LLM代理","可视化可供性","方法比较"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2507.17024","has_summary":true},{"id":"2507.10933","title":"Artificial Finance: How AI Thinks About Money","zh_title":"人工金融：AI如何思考金钱","primary_category":"econ.GN","date":"2025-07-15","score":9,"bucket":"selected","tags":["LLM仿真","金融决策","跨文化对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2507.10933","has_summary":true},{"id":"2507.10342","title":"Using AI to replicate human experimental results: a motion study","zh_title":"使用AI复现人类实验结果：一项运动研究","primary_category":"cs.CL","date":"2025-07-14","score":9,"bucket":"selected","tags":["LLM仿真","人类数据对照","心理语言学"],"rubric_hits":["A1","A2","B1"],"abs_url":"https://arxiv.org/abs/2507.10342","has_summary":true},{"id":"2507.09657","title":"Negotiating Comfort: Simulating Personality-Driven LLM Agents in Shared Residential Social Networks","zh_title":"协商舒适度：在共享住宅社交网络中模拟个性驱动的LLM智能体","primary_category":"cs.SI","date":"2025-07-13","score":5,"bucket":"other","tags":["LLM智能体","社会模拟","个性驱动"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2507.09657","has_summary":true},{"id":"2507.08584","title":"To Trade or Not to Trade: An Agentic Approach to Estimating Market Risk Improves Trading Decisions","zh_title":"交易与否：一种基于智能体的市场风险估计方法改善交易决策","primary_category":"q-fin.ST","date":"2025-07-11","score":0,"bucket":"other","tags":["智能体系统","金融交易","随机微分方程"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2507.08584","has_summary":false},{"id":"2507.07188","title":"Prompt Perturbations Reveal Human-Like Biases in Large Language Model Survey Responses","zh_title":"提示扰动揭示大语言模型调查响应中类人偏差","primary_category":"cs.CL","date":"2025-07-09","score":9,"bucket":"selected","tags":["LLM仿真","调查偏差","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2507.07188","has_summary":true},{"id":"2507.08019","title":"Signal or Noise? Evaluating Large Language Models in Resume Screening Across Contextual Variations and Human Expert Benchmarks","zh_title":"信号还是噪声？评估大语言模型在简历筛选中的表现：情境变化与人类专家基准","primary_category":"cs.CL","date":"2025-07-08","score":7,"bucket":"pending","tags":["LLM仿真","人类对照","招聘决策"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2507.08019","has_summary":true},{"id":"2506.23610","title":"Evaluating the Simulation of Human Personality-Driven Susceptibility to Misinformation with LLMs","zh_title":"评估大语言模型对人类人格驱动的错误信息易感性模拟","primary_category":"cs.CL","date":"2025-06-30","score":9,"bucket":"selected","tags":["LLM仿真","人格与行为","错误信息"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2506.23610","has_summary":true},{"id":"2506.23107","title":"Can Large Language Models Capture Human Risk Preferences? A Cross-Cultural Study","zh_title":"大语言模型能捕捉人类风险偏好吗？一项跨文化研究","primary_category":"cs.AI","date":"2025-06-29","score":9,"bucket":"selected","tags":["LLM仿真","风险偏好","跨文化对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2506.23107","has_summary":true},{"id":"2506.21974","title":"Don't Trust Generative Agents to Mimic Communication on Social Networks Unless You Benchmarked their Empirical Realism","zh_title":"不要相信生成式智能体能模仿社交网络上的交流，除非你对其经验现实主义进行了基准测试","primary_category":"cs.CL","date":"2025-06-27","score":9,"bucket":"selected","tags":["LLM仿真","社交网络","经验现实主义"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2506.21974","has_summary":true},{"id":"2507.02919","title":"ChatGPT is not A Man but Das Man: Representativeness and Structural Consistency of Silicon Samples Generated by Large Language Models","zh_title":"ChatGPT不是人而是常人：大语言模型生成硅样本的代表性与结构一致性","primary_category":"cs.CL","date":"2025-06-25","score":10,"bucket":"selected","tags":["LLM人类仿真","调查数据对照","算法偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2507.02919","has_summary":true},{"id":"2506.15041","title":"Identifying economic narratives in large text corpora -- An integrated approach using Large Language Models","zh_title":"利用大语言模型识别大型文本语料库中的经济叙事——一种集成方法","primary_category":"econ.GN","date":"2025-06-18","score":5,"bucket":"other","tags":["LLM标注","经济叙事","文本分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2506.15041","has_summary":false},{"id":"2506.21587","title":"A Cross-Cultural Comparison of LLM-based Public Opinion Simulation: Evaluating Chinese and U.S. Models on Diverse Societies","zh_title":"基于大语言模型的舆论仿真跨文化比较：评估中美模型在多元社会上的表现","primary_category":"cs.CL","date":"2025-06-17","score":10,"bucket":"selected","tags":["LLM人类仿真","舆论模拟","跨文化比较"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2506.21587","has_summary":true},{"id":"2506.14611","title":"Exploring MLLMs Perception of Network Visualization Principles","zh_title":"探索多模态大语言模型对网络可视化原则的感知","primary_category":"cs.HC","date":"2025-06-17","score":9,"bucket":"selected","tags":["LLM仿真","人类被试替代","网络感知实验"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2506.14611","has_summary":true},{"id":"2506.21574","title":"Digital Gatekeepers: Exploring Large Language Model's Role in Immigration Decisions","zh_title":"数字守门人：探索大语言模型在移民决策中的作用","primary_category":"cs.CL","date":"2025-06-15","score":8,"bucket":"selected","tags":["LLM仿真","决策实验","公平性评估"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2506.21574","has_summary":true},{"id":"2506.12664","title":"Behavioral Generative Agents for Energy Operations","zh_title":"用于能源运营的行为生成式智能体","primary_category":"cs.AI","date":"2025-06-14","score":7,"bucket":"pending","tags":["LLM仿真","能源经济","行为建模"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2506.12664","has_summary":true},{"id":"2506.10546","title":"Nowcasting the euro area with social media data","zh_title":"利用社交媒体数据对欧元区进行即时预测","primary_category":"econ.EM","date":"2025-06-12","score":0,"bucket":"other","tags":["经济预测","社交媒体分析","大语言模型应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2506.10546","has_summary":false},{"id":"2507.19495","title":"Simulating Human Behavior with the Psychological-mechanism Agent: Integrating Feeling, Thought, and Action","zh_title":"基于心理机制代理的人类行为模拟：整合感受、思维与行动","primary_category":"cs.HC","date":"2025-06-04","score":9,"bucket":"selected","tags":["人类行为仿真","心理学实验复现","生成式代理"],"rubric_hits":["A1","A2","A5","B1","B2"],"abs_url":"https://arxiv.org/abs/2507.19495","has_summary":true},{"id":"2506.06377","title":"Evaluating Large Language Model Capabilities in Assessing Spatial Econometrics Research","zh_title":"评估大语言模型在空间计量经济学研究评估中的能力","primary_category":"cs.CY","date":"2025-06-04","score":5,"bucket":"other","tags":["LLM评估","同行评审","空间计量经济学"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2506.06377","has_summary":false},{"id":"2506.04478","title":"Matching Markets Meet LLMs: Algorithmic Reasoning with Ranked Preferences","zh_title":"匹配市场遇见大语言模型：基于排序偏好的算法推理","primary_category":"cs.AI","date":"2025-06-04","score":0,"bucket":"other","tags":["匹配市场","算法推理","LLM评估"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2506.04478","has_summary":false},{"id":"2506.02827","title":"TO-GATE: Clarifying Questions and Summarizing Responses with Trajectory Optimization for Eliciting Human Preference","zh_title":"TO-GATE：通过轨迹优化澄清问题并总结回答以获取人类偏好","primary_category":"cs.CL","date":"2025-06-03","score":0,"bucket":"other","tags":["偏好获取","多轮对话","轨迹优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2506.02827","has_summary":false},{"id":"2506.00856","title":"Can AI Master Econometrics? Evidence from Econometrics AI Agent on Expert-Level Tasks","zh_title":"AI能掌握计量经济学吗？来自专家级任务中计量经济学AI智能体的证据","primary_category":"econ.EM","date":"2025-06-01","score":0,"bucket":"other","tags":["AI智能体","计量经济学","自动化数据分析"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2506.00856","has_summary":false},{"id":"2505.24640","title":"Efficient Text Encoders for Labor Market Analysis","zh_title":"面向劳动力市场分析的高效文本编码器","primary_category":"cs.CL","date":"2025-05-30","score":0,"bucket":"other","tags":["技能抽取","职位标准化","对比学习"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2505.24640","has_summary":false},{"id":"2505.23025","title":"Learning to Regulate: A New Event-Level Dataset of Capital Control Measures","zh_title":"学习监管：一个新的资本管制措施事件级数据集","primary_category":"econ.GN","date":"2025-05-29","score":0,"bucket":"other","tags":["资本管制","事件数据集","LLM信息抽取"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2505.23025","has_summary":false},{"id":"2505.21997","title":"Leveraging Interview-Informed LLMs to Model Survey Responses: Comparative Insights from AI-Generated and Human Data","zh_title":"利用访谈信息引导大语言模型建模调查回答：AI生成数据与人类数据的比较洞察","primary_category":"cs.CL","date":"2025-05-28","score":9,"bucket":"selected","tags":["LLM仿真","调查回答","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2505.21997","has_summary":true},{"id":"2505.21371","title":"When Experimental Economics Meets Large Language Models: Evidence-based Tactics","zh_title":"当实验经济学遇上大语言模型：基于证据的实践策略","primary_category":"econ.GN","date":"2025-05-27","score":7,"bucket":"pending","tags":["LLM实验设计","方法论","实验经济学"],"rubric_hits":["A2","A4","B4"],"abs_url":"https://arxiv.org/abs/2505.21371","has_summary":true},{"id":"2505.17479","title":"Twin-2K-500: A dataset for building digital twins of over 2,000 people based on their answers to over 500 questions","zh_title":"Twin-2K-500：基于2000余人对500余题回答构建数字孪生的数据集","primary_category":"cs.CY","date":"2025-05-23","score":10,"bucket":"selected","tags":["数字孪生","人类仿真","行为经济学"],"rubric_hits":["A1","A2","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2505.17479","has_summary":true},{"id":"2505.16188","title":"SAE-SSV: Supervised Steering in Sparse Representation Spaces for Reliable Control of Language Models","zh_title":"SAE-SSV：稀疏表示空间中的监督引导以实现语言模型的可靠控制","primary_category":"cs.CL","date":"2025-05-22","score":0,"bucket":"other","tags":["模型控制","稀疏自编码器","可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2505.16188","has_summary":false},{"id":"2505.16147","title":"Losing is for Cherishing: Data Valuation Based on Machine Unlearning and Shapley Value","zh_title":"失去是为了珍惜：基于机器遗忘和Shapley值的数据估值","primary_category":"cs.AI","date":"2025-05-22","score":0,"bucket":"other","tags":["数据估值","机器遗忘","Shapley值"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2505.16147","has_summary":false},{"id":"2505.14588","title":"Generative AI at the Crossroads: Light Bulb, Dynamo, or Microscope?","zh_title":"生成式AI的十字路口：灯泡、发电机还是显微镜？","primary_category":"econ.GN","date":"2025-05-20","score":0,"bucket":"other","tags":["生成式AI","生产率","技术分类"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2505.14588","has_summary":false},{"id":"2505.12923","title":"The Traitors: Deception and Trust in Multi-Agent Language Model Simulations","zh_title":"《叛徒》：多智能体语言模型仿真中的欺骗与信任","primary_category":"cs.AI","date":"2025-05-19","score":5,"bucket":"other","tags":["多智能体仿真","欺骗检测","社会推理"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2505.12923","has_summary":true},{"id":"2505.10309","title":"A large-scale evaluation of commonsense knowledge in humans and large language models","zh_title":"人类与大语言模型常识知识的大规模评估","primary_category":"cs.AI","date":"2025-05-15","score":9,"bucket":"selected","tags":["LLM仿真","人类对照","常识知识"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2505.10309","has_summary":true},{"id":"2505.09938","title":"Design and Evaluation of Generative Agent-based Platform for Human-Assistant Interaction Research: A Tale of 10 User Studies","zh_title":"基于生成式智能体的仿真平台设计与评估：10项用户研究的故事","primary_category":"cs.HC","date":"2025-05-15","score":7,"bucket":"pending","tags":["LLM仿真","人机交互","用户研究复现"],"rubric_hits":["A1","A2","B1"],"abs_url":"https://arxiv.org/abs/2505.09938","has_summary":true},{"id":"2505.09396","title":"The Influence of Human-inspired Agentic Sophistication in LLM-driven Strategic Reasoners","zh_title":"人类启发的智能体复杂度对LLM驱动战略推理者的影响","primary_category":"cs.AI","date":"2025-05-14","score":8,"bucket":"selected","tags":["LLM仿真","博弈实验","人类对照"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2505.09396","has_summary":true},{"id":"2505.07457","title":"Can Generative AI agents behave like humans? Evidence from laboratory market experiments","zh_title":"生成式AI智能体能像人类一样行为吗？来自实验室市场实验的证据","primary_category":"econ.GN","date":"2025-05-12","score":9,"bucket":"selected","tags":["LLM仿真","市场实验","人类行为对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2505.07457","has_summary":true},{"id":"2505.07653","title":"JobHop: A Large-Scale Dataset of Career Trajectories","zh_title":"JobHop：一个大规模职业轨迹数据集","primary_category":"cs.CL","date":"2025-05-12","score":0,"bucket":"other","tags":["信息抽取","劳动力市场","数据集"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2505.07653","has_summary":false},{"id":"2505.06702","title":"Do Language Model Agents Align with Humans in Rating Visualizations? An Empirical Study","zh_title":"语言模型代理在可视化评分中与人类对齐吗？一项实证研究","primary_category":"cs.HC","date":"2025-05-10","score":9,"bucket":"selected","tags":["LLM仿真","人类数据对照","可视化评估"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2505.06702","has_summary":true},{"id":"2507.18639","title":"People Are Highly Cooperative with Large Language Models, Especially When Communication Is Possible or Following Human Interaction","zh_title":"人们与大型语言模型高度合作，尤其在可沟通或继人类互动之后","primary_category":"cs.HC","date":"2025-05-10","score":9,"bucket":"selected","tags":["LLM仿真","行为博弈","人机合作"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2507.18639","has_summary":true},{"id":"2505.05863","title":"Evolutionary ecology of words","zh_title":"词汇的进化生态学","primary_category":"q-bio.PE","date":"2025-05-09","score":0,"bucket":"other","tags":["多智能体系统","进化博弈","语言模型"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2505.05863","has_summary":false},{"id":"2505.00036","title":"A Framework to Assess the Persuasion Risks Large Language Model Chatbots Pose to Democratic Societies","zh_title":"评估大语言模型聊天机器人对民主社会说服风险的框架","primary_category":"cs.CL","date":"2025-04-29","score":9,"bucket":"selected","tags":["LLM仿真","政治说服","人类数据对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2505.00036","has_summary":true},{"id":"2504.20628","title":"Cognitive maps are generative programs","zh_title":"认知地图是生成式程序","primary_category":"cs.AI","date":"2025-04-29","score":5,"bucket":"other","tags":["认知建模","人类行为预测","LLM先验嵌入"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2504.20628","has_summary":true},{"id":"2504.17993","title":"Improving Language Model Personas via Rationalization with Psychological Scaffolds","zh_title":"通过心理支架合理化改进语言模型角色","primary_category":"cs.CL","date":"2025-04-25","score":7,"bucket":"pending","tags":["LLM角色仿真","人类偏好预测","心理支架"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2504.17993","has_summary":true},{"id":"2504.19940","title":"Assessing the Potential of Generative Agents in Crowdsourced Fact-Checking","zh_title":"评估生成式智能体在众包事实核查中的潜力","primary_category":"cs.CL","date":"2025-04-24","score":9,"bucket":"selected","tags":["LLM仿真","众包事实核查","人类行为对照"],"rubric_hits":["A1","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2504.19940","has_summary":true},{"id":"2504.11671","title":"Computational Basis of LLM's Decision Making in Social Simulation","zh_title":"LLM在社会仿真中决策的计算基础","primary_category":"cs.AI","date":"2025-04-16","score":7,"bucket":"pending","tags":["LLM社会仿真","独裁者博弈","表征操控"],"rubric_hits":["A1","A3","B2"],"abs_url":"https://arxiv.org/abs/2504.11671","has_summary":true},{"id":"2504.09663","title":"Ordinary Least Squares as an Attention Mechanism","zh_title":"普通最小二乘法作为一种注意力机制","primary_category":"cs.LG","date":"2025-04-13","score":0,"bucket":"other","tags":["注意力机制","OLS","统计方法"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2504.09663","has_summary":false},{"id":"2504.09059","title":"Large Language Models integration in Smart Grids","zh_title":"大语言模型在智能电网中的集成","primary_category":"cs.CY","date":"2025-04-12","score":0,"bucket":"other","tags":["智能电网","LLM应用","多智能体系统"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2504.09059","has_summary":false},{"id":"2504.08260","title":"Evaluating the Bias in LLMs for Surveying Opinion and Decision Making in Healthcare","zh_title":"评估大语言模型在医疗意见与决策调查中的偏差","primary_category":"cs.CL","date":"2025-04-11","score":10,"bucket":"selected","tags":["LLM仿真","人类数据对照","医疗决策偏差"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2504.08260","has_summary":true},{"id":"2504.13908","title":"AI-Assisted Conversational Interviewing: Effects on Data Quality and Respondent Experience","zh_title":"AI辅助的对话式访谈：对数据质量和受访者体验的影响","primary_category":"cs.HC","date":"2025-04-09","score":7,"bucket":"pending","tags":["AI辅助调查","对话式访谈","数据质量"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2504.13908","has_summary":false},{"id":"2504.05862","title":"Are Generative AI Agents Effective Personalized Financial Advisors?","zh_title":"生成式AI代理能成为有效的个性化理财顾问吗？","primary_category":"cs.AI","date":"2025-04-08","score":8,"bucket":"selected","tags":["LLM仿真","人机对照实验","金融行为"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2504.05862","has_summary":true},{"id":"2504.01566","title":"GPT Adoption and the Impact of Disclosure Policies","zh_title":"GPT采用与披露政策的影响","primary_category":"econ.GN","date":"2025-04-02","score":5,"bucket":"other","tags":["LLM标注","调查实验","代理理论"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2504.01566","has_summary":false},{"id":"2503.22726","title":"InfoBid: A Simulation Framework for Studying Information Disclosure in Auctions with Large Language Model-based Agents","zh_title":"InfoBid：基于大语言模型代理研究拍卖中信息披露的仿真框架","primary_category":"cs.GT","date":"2025-03-26","score":8,"bucket":"selected","tags":["LLM仿真","拍卖实验","经济行为"],"rubric_hits":["A3","B2","B4"],"abs_url":"https://arxiv.org/abs/2503.22726","has_summary":true},{"id":"2503.12556","title":"From Guessing to Asking: An Approach to Resolving the Persona Knowledge Gap in LLMs during Multi-Turn Conversations","zh_title":"从猜测到询问：解决多轮对话中大语言模型角色知识缺口的方法","primary_category":"cs.CL","date":"2025-03-16","score":0,"bucket":"other","tags":["对话系统","个性化推荐","角色扮演"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2503.12556","has_summary":false},{"id":"2503.11531","title":"Potential of large language model-powered nudges for promoting daily water and energy conservation","zh_title":"大语言模型驱动的助推在促进日常节水节能中的潜力","primary_category":"cs.CY","date":"2025-03-14","score":9,"bucket":"selected","tags":["LLM仿真","行为干预","节能实验"],"rubric_hits":["A1","A2","B1","B2"],"abs_url":"https://arxiv.org/abs/2503.11531","has_summary":true},{"id":"2503.10990","title":"Statistical Impossibility and Possibility of Aligning LLMs with Human Preferences: From Condorcet Paradox to Nash Equilibrium","zh_title":"将大语言模型与人类偏好对齐的统计不可能性与可能性：从孔多塞悖论到纳什均衡","primary_category":"cs.GT","date":"2025-03-14","score":0,"bucket":"other","tags":["偏好对齐","统计极限","RLHF"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2503.10990","has_summary":false},{"id":"2503.10248","title":"LLM Agents Display Human Biases but Exhibit Distinct Learning Patterns","zh_title":"LLM智能体表现出人类偏见但学习模式不同","primary_category":"cs.AI","date":"2025-03-13","score":9,"bucket":"selected","tags":["LLM仿真","决策实验","人类对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2503.10248","has_summary":true},{"id":"2503.09639","title":"Can A Society of Generative Agents Simulate Human Behavior and Inform Public Health Policy? A Case Study on Vaccine Hesitancy","zh_title":"生成式智能体社会能否模拟人类行为并为公共卫生政策提供信息？以疫苗犹豫为例","primary_category":"cs.MA","date":"2025-03-12","score":9,"bucket":"selected","tags":["LLM人类仿真","疫苗犹豫","政策评估"],"rubric_hits":["A1","A3","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2503.09639","has_summary":true},{"id":"2503.07510","title":"Sometimes the Model doth Preach: Quantifying Religious Bias in Open LLMs through Demographic Analysis in Asian Nations","zh_title":"有时模型在布道：通过亚洲国家人口统计分析量化开放LLM中的宗教偏见","primary_category":"cs.CY","date":"2025-03-10","score":8,"bucket":"selected","tags":["LLM仿真","宗教偏见","人类数据对照"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2503.07510","has_summary":true},{"id":"2503.05529","title":"PoSSUM: A Protocol for Surveying Social-media Users with Multimodal LLMs","zh_title":"PoSSUM：一种利用多模态大语言模型调查社交媒体用户的协议","primary_category":"stat.AP","date":"2025-03-07","score":9,"bucket":"selected","tags":["LLM仿真","选举预测","人类行为对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2503.05529","has_summary":true},{"id":"2503.05481","title":"Maximum Hallucination Standards for Domain-Specific Large Language Models","zh_title":"领域特定大语言模型的最大幻觉标准","primary_category":"econ.GN","date":"2025-03-07","score":0,"bucket":"other","tags":["LLM幻觉","经济学模型","产品属性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2503.05481","has_summary":false},{"id":"2503.04804","title":"What do Large Language Models Say About Animals? Investigating Risks of Animal Harm in Generated Text","zh_title":"大型语言模型对动物有何看法？探究生成文本中动物伤害的风险","primary_category":"cs.CY","date":"2025-03-03","score":0,"bucket":"other","tags":["LLM安全评测","动物伦理","基准数据集"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2503.04804","has_summary":false},{"id":"2503.00725","title":"Causal Inference on Outcomes Learned from Text","zh_title":"基于文本学习结果的因果推断","primary_category":"econ.EM","date":"2025-03-02","score":5,"bucket":"other","tags":["因果推断","文本分析","LLM标注"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2503.00725","has_summary":false},{"id":"2503.00177","title":"Steering Large Language Model Activations in Sparse Spaces","zh_title":"在稀疏空间中操控大语言模型激活","primary_category":"cs.LG","date":"2025-02-28","score":0,"bucket":"other","tags":["激活操控","稀疏自编码器","AI对齐"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2503.00177","has_summary":false},{"id":"2502.18371","title":"MindMem: Multimodal for Predicting Advertisement Memorability Using LLMs and Deep Learning","zh_title":"MindMem：利用LLM和深度学习预测广告记忆度的多模态模型","primary_category":"cs.AI","date":"2025-02-25","score":4,"bucket":"other","tags":["广告记忆度","多模态预测","LLM仿真"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2502.18371","has_summary":false},{"id":"2502.16280","title":"Human Preferences in Large Language Model Latent Space: A Technical Analysis on the Reliability of Synthetic Data in Voting Outcome Prediction","zh_title":"大语言模型潜在空间中的人类偏好：合成数据在投票结果预测中可靠性的技术分析","primary_category":"cs.LG","date":"2025-02-22","score":9,"bucket":"selected","tags":["LLM人类仿真","合成数据可靠性","政治偏好预测"],"rubric_hits":["A1","A2","A4","B1","B3","B4"],"abs_url":"https://arxiv.org/abs/2502.16280","has_summary":true},{"id":"2502.14499","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","zh_title":"MLGym：推进AI研究智能体的新框架与基准","primary_category":"cs.CL","date":"2025-02-20","score":0,"bucket":"other","tags":["AI研究智能体","多智能体协作","基准测试"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2502.14499","has_summary":false},{"id":"2502.15800","title":"LLM Agents Do Not Replicate Human Market Traders: Evidence From Experimental Finance","zh_title":"LLM代理无法复现人类市场交易者：来自实验金融的证据","primary_category":"q-fin.TR","date":"2025-02-18","score":10,"bucket":"selected","tags":["LLM仿真","实验金融","人类行为对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2502.15800","has_summary":true},{"id":"2502.10266","title":"Are Large Language Models the future crowd workers of Linguistics?","zh_title":"大语言模型能否成为语言学未来的众包工作者？","primary_category":"cs.CL","date":"2025-02-14","score":7,"bucket":"pending","tags":["LLM仿真","人类数据对照","语言学实验"],"rubric_hits":["A1","B1","B4"],"abs_url":"https://arxiv.org/abs/2502.10266","has_summary":true},{"id":"2502.10308","title":"LLM-Powered Preference Elicitation in Combinatorial Assignment","zh_title":"基于大语言模型的组合分配偏好获取","primary_category":"cs.AI","date":"2025-02-14","score":7,"bucket":"pending","tags":["LLM代理","偏好获取","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2502.10308","has_summary":true},{"id":"2503.05708","title":"On Large Language Models as Data Sources for Policy Deliberation on Climate Change and Sustainability","zh_title":"大语言模型作为气候与可持续性政策审议数据源的研究","primary_category":"cs.CY","date":"2025-02-13","score":7,"bucket":"pending","tags":["LLM仿真","政策评估","人类数据对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2503.05708","has_summary":true},{"id":"2502.08691","title":"AgentSociety: Large-Scale Simulation of LLM-Driven Generative Agents Advances Understanding of Human Behaviors and Society","zh_title":"AgentSociety：大规模LLM驱动生成式代理模拟推进对人类行为与社会的理解","primary_category":"cs.SI","date":"2025-02-12","score":10,"bucket":"selected","tags":["LLM社会仿真","人类行为复现","政策评估"],"rubric_hits":["A1","A3","A5","B1","B2","B3"],"abs_url":"https://arxiv.org/abs/2502.08691","has_summary":true},{"id":"2502.07307","title":"CreAgent: Towards Long-Term Evaluation of Recommender System under Platform-Creator Information Asymmetry","zh_title":"CreAgent：平台-创作者信息不对称下推荐系统长期评估","primary_category":"cs.IR","date":"2025-02-11","score":5,"bucket":"other","tags":["LLM智能体","推荐系统仿真","创作者行为模拟"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2502.07307","has_summary":true},{"id":"2502.07736","title":"Menu Pricing of Large Language Models","zh_title":"大语言模型的菜单定价","primary_category":"econ.TH","date":"2025-02-11","score":0,"bucket":"other","tags":["定价策略","经济学理论","产品设计"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2502.07736","has_summary":false},{"id":"2502.06387","title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","zh_title":"人类如何帮助大语言模型：评估与激励人类偏好标注员","primary_category":"cs.LG","date":"2025-02-10","score":0,"bucket":"other","tags":["人类标注","偏好对齐","激励设计"],"rubric_hits":["C5"],"abs_url":"https://arxiv.org/abs/2502.06387","has_summary":false},{"id":"2502.03158","title":"Strategizing with AI: Insights from a Beauty Contest Experiment","zh_title":"与AI博弈：选美竞赛实验的启示","primary_category":"econ.GN","date":"2025-02-05","score":9,"bucket":"selected","tags":["LLM仿真","行为博弈","人类数据对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2502.03158","has_summary":true},{"id":"2502.00070","title":"Can AI Solve the Peer Review Crisis? A Large Scale Cross Model Experiment of LLMs' Performance and Biases in Evaluating over 1000 Economics Papers","zh_title":"AI能解决同行评审危机吗？一项关于LLM在评估1000多篇经济学论文中的表现与偏差的大规模跨模型实验","primary_category":"cs.CY","date":"2025-01-31","score":5,"bucket":"other","tags":["LLM审稿","经济学论文评估","偏差分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2502.00070","has_summary":true},{"id":"2501.19266","title":"Jackpot! Alignment as a Maximal Lottery","zh_title":"头奖！对齐作为最大彩票","primary_category":"cs.AI","date":"2025-01-31","score":4,"bucket":"other","tags":["RLHF","社会选择","多智能体对齐"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2501.19266","has_summary":false},{"id":"2501.17310","title":"Probing LLM World Models: Enhancing Guesstimation with Wisdom of Crowds Decoding","zh_title":"探究LLM世界模型：用群体智慧解码增强估算能力","primary_category":"cs.AI","date":"2025-01-28","score":7,"bucket":"pending","tags":["LLM仿真","群体智慧","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2501.17310","has_summary":true},{"id":"2501.14294","title":"Examining Alignment of Large Language Models through Representative Heuristics: The Case of Political Stereotypes","zh_title":"通过代表性启发式检验大语言模型的对齐：以政治刻板印象为例","primary_category":"cs.CL","date":"2025-01-24","score":9,"bucket":"selected","tags":["LLM人类仿真","政治态度模拟","代表性启发式"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2501.14294","has_summary":true},{"id":"2501.13955","title":"Guided Persona-based AI Surveys: Can we replicate personal mobility preferences at scale using LLMs?","zh_title":"基于引导式角色的AI调查：能否利用LLM大规模复制个人出行偏好？","primary_category":"cs.CL","date":"2025-01-20","score":9,"bucket":"selected","tags":["LLM人类仿真","合成调查数据","出行偏好"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2501.13955","has_summary":true},{"id":"2501.08579","title":"LLM-based Human Simulations Have Not Yet Been Reliable","zh_title":"基于大语言模型的人类仿真尚未可靠","primary_category":"cs.CL","date":"2025-01-15","score":9,"bucket":"selected","tags":["LLM人类仿真","可靠性评估","方法论框架"],"rubric_hits":["A1","A2","A4","B1","B4"],"abs_url":"https://arxiv.org/abs/2501.08579","has_summary":true},{"id":"2501.07663","title":"Enhancing Talent Employment Insights Through Feature Extraction with LLM Finetuning","zh_title":"通过LLM微调的特征提取增强人才就业洞察","primary_category":"cs.CL","date":"2025-01-13","score":0,"bucket":"other","tags":["NLP","信息抽取","劳动力市场分析"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2501.07663","has_summary":false},{"id":"2501.06834","title":"LLMs Model Non-WEIRD Populations: Experiments with Synthetic Cultural Agents","zh_title":"LLM模拟非WEIRD人群：合成文化代理实验","primary_category":"cs.AI","date":"2025-01-12","score":9,"bucket":"selected","tags":["LLM仿真","行为实验","跨文化经济研究"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2501.06834","has_summary":true},{"id":"2501.00745","title":"Dynamics of Adversarial Attacks on Large Language Model-Based Search Engines","zh_title":"基于大语言模型的搜索引擎对抗攻击动力学","primary_category":"cs.CL","date":"2025-01-01","score":0,"bucket":"other","tags":["对抗攻击","博弈论","搜索引擎"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2501.00745","has_summary":false},{"id":"2412.19363","title":"Large Language Models for Market Research: A Data-augmentation Approach","zh_title":"用于市场研究的大语言模型：一种数据增强方法","primary_category":"cs.AI","date":"2024-12-26","score":8,"bucket":"selected","tags":["LLM仿真","联合分析","数据增强"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2412.19363","has_summary":true},{"id":"2412.19245","title":"Sentiment trading with large language models","zh_title":"基于大语言模型的情感交易研究","primary_category":"q-fin.CP","date":"2024-12-26","score":0,"bucket":"other","tags":["情感分析","金融预测","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2412.19245","has_summary":false},{"id":"2412.18497","title":"Neuron-Level Differentiation of Memorization and Generalization in Large Language Models","zh_title":"大型语言模型中记忆与泛化的神经元级分化","primary_category":"cs.CL","date":"2024-12-24","score":0,"bucket":"other","tags":["神经元分析","记忆与泛化","模型可解释性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2412.18497","has_summary":false},{"id":"2412.18061","title":"Lla-VAP: LSTM Ensemble of Llama and VAP for Turn-Taking Prediction","zh_title":"Lla-VAP：用于轮次预测的 Llama 与 VAP 的 LSTM 集成","primary_category":"cs.SD","date":"2024-12-24","score":0,"bucket":"other","tags":["对话系统","轮次预测","多模态集成"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2412.18061","has_summary":false},{"id":"2412.16265","title":"Autoware.Flex: Human-Instructed Dynamically Reconfigurable Autonomous Driving Systems","zh_title":"Autoware.Flex：人类指令驱动的动态可重构自动驾驶系统","primary_category":"cs.AI","date":"2024-12-20","score":0,"bucket":"other","tags":["自动驾驶","人机交互","LLM指令翻译"],"rubric_hits":["C2"],"abs_url":"https://arxiv.org/abs/2412.16265","has_summary":false},{"id":"2412.14161","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","zh_title":"TheAgentCompany：在重要现实世界任务上评测LLM智能体","primary_category":"cs.CL","date":"2024-12-18","score":0,"bucket":"other","tags":["LLM智能体","基准测试","任务自动化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2412.14161","has_summary":false},{"id":"2412.10635","title":"Do LLMs Act as Repositories of Causal Knowledge?","zh_title":"大语言模型是否充当因果知识库？","primary_category":"econ.EM","date":"2024-12-14","score":0,"bucket":"other","tags":["因果推断","LLM评测","混杂因子识别"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2412.10635","has_summary":false},{"id":"2412.07031","title":"Large Language Models: An Applied Econometric Framework","zh_title":"大语言模型：一个应用计量经济学框架","primary_category":"econ.EM","date":"2024-12-09","score":5,"bucket":"other","tags":["LLM标注","计量方法","文本分析"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2412.07031","has_summary":true},{"id":"2412.02065","title":"Leveraging Large Language Models to Democratize Access to Costly Datasets for Academic Research","zh_title":"利用大语言模型使昂贵数据集获取民主化以促进学术研究","primary_category":"q-fin.GN","date":"2024-12-03","score":0,"bucket":"other","tags":["LLM数据提取","信息抽取","研究资源民主化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2412.02065","has_summary":false},{"id":"2412.01069","title":"The Promise and Peril of Generative AI: Evidence from GPT as Sell-Side Analysts","zh_title":"生成式AI的前景与风险：来自GPT作为卖方分析师的证据","primary_category":"q-fin.GN","date":"2024-12-02","score":7,"bucket":"pending","tags":["LLM仿真","金融预测","人类对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2412.01069","has_summary":false},{"id":"2411.11581","title":"OASIS: Open Agent Social Interaction Simulations with One Million Agents","zh_title":"OASIS：百万代理的开放代理社交互动模拟","primary_category":"cs.CL","date":"2024-11-18","score":7,"bucket":"pending","tags":["社会模拟","LLM代理","信息传播"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2411.11581","has_summary":false},{"id":"2411.06790","title":"Large-scale moral machine experiment on large language models","zh_title":"基于大语言模型的大规模道德机器实验","primary_category":"cs.CY","date":"2024-11-11","score":5,"bucket":"other","tags":["LLM道德判断","自动驾驶伦理","人类对齐"],"rubric_hits":["D2","C2"],"abs_url":"https://arxiv.org/abs/2411.06790","has_summary":true},{"id":"2411.05194","title":"Interactive Dialogue Agents via Reinforcement Learning on Hindsight Regenerations","zh_title":"基于事后重写的强化学习交互式对话代理","primary_category":"cs.LG","date":"2024-11-07","score":0,"bucket":"other","tags":["对话代理","强化学习","说服任务"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2411.05194","has_summary":false},{"id":"2410.22203","title":"Democratizing Reward Design for Personal and Representative Value-Alignment","zh_title":"个性化与代表性价值对齐的奖励设计民主化","primary_category":"cs.AI","date":"2024-10-29","score":0,"bucket":"other","tags":["价值对齐","个性化奖励模型","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2410.22203","has_summary":false},{"id":"2410.19599","title":"Take Caution in Using LLMs as Human Surrogates: Scylla Ex Machina","zh_title":"谨慎使用LLM作为人类替代品：Scylla Ex Machina","primary_category":"econ.GN","date":"2024-10-25","score":10,"bucket":"selected","tags":["LLM人类仿真","行为博弈","算法保真度"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2410.19599","has_summary":true},{"id":"2410.10665","title":"Double Jeopardy and Climate Impact in the Use of Large Language Models: Socio-economic Disparities and Reduced Utility for Non-English Speakers","zh_title":"使用大语言模型的双重困境与气候影响：非英语使用者的社会经济差距与效用降低","primary_category":"cs.CL","date":"2024-10-14","score":0,"bucket":"other","tags":["LLM公平性","语言资源不均","NLP评测"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2410.10665","has_summary":false},{"id":"2409.19430","title":"'Simulacrum of Stories': Examining Large Language Models as Qualitative Research Participants","zh_title":"“故事的拟像”：审视大语言模型作为定性研究参与者","primary_category":"cs.HC","date":"2024-09-28","score":9,"bucket":"selected","tags":["LLM仿真","定性研究","方法论批判"],"rubric_hits":["A1","A4","B4"],"abs_url":"https://arxiv.org/abs/2409.19430","has_summary":true},{"id":"2409.14202","title":"Mining Causality: AI-Assisted Search for Instrumental Variables","zh_title":"挖掘因果关系：人工智能辅助的工具变量搜索","primary_category":"econ.EM","date":"2024-09-21","score":5,"bucket":"other","tags":["工具变量","LLM模拟","因果推断"],"rubric_hits":["D3"],"abs_url":"https://arxiv.org/abs/2409.14202","has_summary":true},{"id":"2409.18988","title":"A Unified Framework to Classify Business Activities into International Standard Industrial Classification through Large Language Models for Circular Economy","zh_title":"通过大语言模型将商业活动分类到国际标准产业分类的统一框架以促进循环经济","primary_category":"cs.CL","date":"2024-09-17","score":0,"bucket":"other","tags":["文本分类","循环经济","LLM应用"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2409.18988","has_summary":false},{"id":"2409.10750","title":"GPT takes the SAT: Tracing changes in Test Difficulty and Math Performance of Students","zh_title":"GPT参加SAT：追踪试题难度与学生数学表现的变化","primary_category":"econ.EM","date":"2024-09-16","score":7,"bucket":"pending","tags":["LLM仿真","教育评估","人类数据对照"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2409.10750","has_summary":true},{"id":"2409.08357","title":"An Experimental Study of Competitive Market Behavior Through LLMs","zh_title":"通过大语言模型对竞争市场行为的实验研究","primary_category":"cs.HC","date":"2024-09-12","score":7,"bucket":"pending","tags":["LLM仿真","市场实验","行为经济学"],"rubric_hits":["A3","B2"],"abs_url":"https://arxiv.org/abs/2409.08357","has_summary":true},{"id":"2409.00128","title":"Can Large Language Models Replace Human Subjects? A Large-Scale Replication of Scenario-Based Experiments in Psychology and Management","zh_title":"大语言模型能替代人类被试吗？心理学与管理学场景实验的大规模复现","primary_category":"cs.CL","date":"2024-08-29","score":10,"bucket":"selected","tags":["LLM仿真","人类被试替代","心理学实验复现"],"rubric_hits":["A1","A2","A5","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2409.00128","has_summary":true},{"id":"2408.05328","title":"From Text to Insight: Leveraging Large Language Models for Performance Evaluation in Management","zh_title":"从文本到洞察：利用大语言模型进行管理中的绩效评估","primary_category":"cs.CL","date":"2024-08-09","score":5,"bucket":"other","tags":["LLM评估","绩效评估","人工标注替代"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2408.05328","has_summary":false},{"id":"2407.04467","title":"Are Large Language Models Strategic Decision Makers? A Study of Performance and Bias in Two-Player Non-Zero-Sum Games","zh_title":"大语言模型是战略决策者吗？双人非零和博弈中的表现与偏差研究","primary_category":"cs.AI","date":"2024-07-05","score":8,"bucket":"selected","tags":["LLM仿真","博弈论","决策偏差"],"rubric_hits":["A1","A2","B2","B4"],"abs_url":"https://arxiv.org/abs/2407.04467","has_summary":true},{"id":"2407.03859","title":"Anthropocentric bias in language model evaluation","zh_title":"语言模型评估中的人类中心偏见","primary_category":"cs.CL","date":"2024-07-04","score":0,"bucket":"other","tags":["LLM评估","认知偏见","方法论"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2407.03859","has_summary":false},{"id":"2407.12032","title":"Large Language Models for Behavioral Economics: Internal Validity and Elicitation of Mental Models","zh_title":"大语言模型用于行为经济学：内部效度与心智模型的引出","primary_category":"cs.HC","date":"2024-06-30","score":7,"bucket":"pending","tags":["LLM仿真","行为经济学","内部效度"],"rubric_hits":["A1","A2","B2"],"abs_url":"https://arxiv.org/abs/2407.12032","has_summary":true},{"id":"2406.19317","title":"Jump Starting Bandits with LLM-Generated Prior Knowledge","zh_title":"用LLM生成的先验知识启动Bandit算法","primary_category":"cs.LG","date":"2024-06-27","score":7,"bucket":"pending","tags":["LLM仿真","人类偏好模拟","上下文Bandit"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2406.19317","has_summary":true},{"id":"2406.17972","title":"LABOR-LLM: Language-Based Occupational Representations with Large Language Models","zh_title":"LABOR-LLM：基于大语言模型的职业表征","primary_category":"cs.LG","date":"2024-06-25","score":0,"bucket":"other","tags":["职业预测","LLM微调","序列建模"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2406.17972","has_summary":false},{"id":"2406.16510","title":"Large Language Models in Student Assessment: Comparing ChatGPT and Human Graders","zh_title":"学生评估中的大语言模型：比较ChatGPT与人类评分员","primary_category":"econ.GN","date":"2024-06-24","score":5,"bucket":"other","tags":["LLM评分","教育评估","人机对比"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2406.16510","has_summary":false},{"id":"2406.14508","title":"Evidence of a log scaling law for political persuasion with large language models","zh_title":"大语言模型政治说服力的对数缩放定律证据","primary_category":"cs.CL","date":"2024-06-20","score":9,"bucket":"selected","tags":["LLM仿真","政治说服","人类数据对照"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2406.14508","has_summary":true},{"id":"2406.13605","title":"Nicer Than Humans: How do Large Language Models Behave in the Prisoner's Dilemma?","zh_title":"比人类更友善：大语言模型在囚徒困境中的行为研究","primary_category":"cs.CY","date":"2024-06-19","score":9,"bucket":"selected","tags":["LLM仿真","囚徒困境","行为博弈"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2406.13605","has_summary":true},{"id":"2406.13558","title":"Enhancing Travel Choice Modeling with Large Language Models: A Prompt-Learning Approach","zh_title":"利用大语言模型增强出行选择建模：一种提示学习方法","primary_category":"cs.AI","date":"2024-06-19","score":7,"bucket":"pending","tags":["出行行为预测","LLM仿真","选择建模"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2406.13558","has_summary":true},{"id":"2406.11426","title":"Can AI with High Reasoning Ability Replicate Human-like Decision Making in Economic Experiments?","zh_title":"高推理能力AI能否复制经济实验中的人类决策？","primary_category":"cs.GT","date":"2024-06-17","score":9,"bucket":"selected","tags":["LLM仿真","经济实验","人类行为对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2406.11426","has_summary":true},{"id":"2406.05972","title":"Decision-Making Behavior Evaluation Framework for LLMs under Uncertain Context","zh_title":"不确定情境下大语言模型决策行为评估框架","primary_category":"cs.AI","date":"2024-06-10","score":7,"bucket":"pending","tags":["LLM决策行为","行为经济学","人类对照"],"rubric_hits":["A1","A2","B2","B4"],"abs_url":"https://arxiv.org/abs/2406.05972","has_summary":true},{"id":"2406.03299","title":"The Good, the Bad, and the Hulk-like GPT: Analyzing Emotional Decisions of Large Language Models in Cooperation and Bargaining Games","zh_title":"好、坏与浩克般的GPT：分析大语言模型在合作与讨价还价博弈中的情绪决策","primary_category":"cs.AI","date":"2024-06-05","score":9,"bucket":"selected","tags":["LLM人类仿真","行为博弈","情绪决策"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2406.03299","has_summary":true},{"id":"2406.01407","title":"Utilizing Large Language Models for Automating Technical Customer Support","zh_title":"利用大语言模型实现技术客户支持自动化","primary_category":"econ.GN","date":"2024-06-03","score":0,"bucket":"other","tags":["LLM应用","客户支持","自动化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2406.01407","has_summary":false},{"id":"2405.19578","title":"The Accuracy of Domain Specific and Descriptive Analysis Generated by Large Language Models","zh_title":"大语言模型生成的领域特定与描述性分析的准确性","primary_category":"cs.CE","date":"2024-05-30","score":5,"bucket":"other","tags":["LLM数据分析","人类判断对比","描述性统计"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2405.19578","has_summary":false},{"id":"2405.19313","title":"Language Models Trained to do Arithmetic Predict Human Risky and Intertemporal Choice","zh_title":"训练做算术的语言模型预测人类风险与跨期选择","primary_category":"cs.AI","date":"2024-05-29","score":8,"bucket":"selected","tags":["LLM仿真","决策行为","认知模型"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2405.19313","has_summary":true},{"id":"2405.09161","title":"Exploring the Potential of Large Language Models for Automation in Technical Customer Service","zh_title":"探索大语言模型在技术客户服务自动化中的潜力","primary_category":"econ.GN","date":"2024-05-15","score":0,"bucket":"other","tags":["LLM应用","客户服务自动化","认知任务"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2405.09161","has_summary":false},{"id":"2405.00981","title":"Bayesian Optimization with LLM-Based Acquisition Functions for Natural Language Preference Elicitation","zh_title":"基于LLM采集函数的贝叶斯优化用于自然语言偏好获取","primary_category":"cs.AI","date":"2024-05-02","score":0,"bucket":"other","tags":["偏好获取","对话推荐","贝叶斯优化"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2405.00981","has_summary":false},{"id":"2404.17143","title":"Quantifying Memorization and Detecting Training Data of Pre-trained Language Models using Japanese Newspaper","zh_title":"使用日本报纸量化预训练语言模型的记忆与检测训练数据","primary_category":"cs.CL","date":"2024-04-26","score":0,"bucket":"other","tags":["预训练语言模型","训练数据记忆","成员推断攻击"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2404.17143","has_summary":false},{"id":"2404.09699","title":"Generative AI for Game Theory-based Mobile Networking","zh_title":"基于博弈论的移动网络生成式人工智能","primary_category":"cs.GT","date":"2024-04-15","score":0,"bucket":"other","tags":["多智能体系统","博弈论","移动网络优化"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2404.09699","has_summary":false},{"id":"2404.08816","title":"Measuring the Quality of Answers in Political Q&As with Large Language Models","zh_title":"用大语言模型衡量政治问答环节中答案的质量","primary_category":"cs.CL","date":"2024-04-12","score":0,"bucket":"other","tags":["NLP评测","政治文本分析","语义相关性"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2404.08816","has_summary":false},{"id":"2404.00530","title":"Comparing Bad Apples to Good Oranges: Aligning Large Language Models via Joint Preference Optimization","zh_title":"比较坏苹果与好橙子：通过联合偏好优化对齐大语言模型","primary_category":"cs.CL","date":"2024-03-31","score":0,"bucket":"other","tags":["偏好优化","LLM对齐","指令微调"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2404.00530","has_summary":false},{"id":"2404.01332","title":"Explaining Large Language Models Decisions Using Shapley Values","zh_title":"使用Shapley值解释大语言模型决策","primary_category":"cs.CL","date":"2024-03-29","score":8,"bucket":"selected","tags":["LLM仿真","Shapley值","认知偏差"],"rubric_hits":["A1","A2","B1","B4"],"abs_url":"https://arxiv.org/abs/2404.01332","has_summary":true},{"id":"2403.16843","title":"Do LLM Agents Have Regret? A Case Study in Online Learning and Games","zh_title":"LLM智能体有遗憾吗？在线学习与博弈案例研究","primary_category":"cs.LG","date":"2024-03-25","score":4,"bucket":"other","tags":["多智能体","在线学习","博弈论"],"rubric_hits":["C1"],"abs_url":"https://arxiv.org/abs/2403.16843","has_summary":false},{"id":"2403.15281","title":"Measuring Gender and Racial Biases in Large Language Models","zh_title":"测量大语言模型中的性别与种族偏见","primary_category":"econ.GN","date":"2024-03-22","score":9,"bucket":"selected","tags":["LLM仿真","偏见测量","劳动力市场"],"rubric_hits":["A1","A2","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2403.15281","has_summary":true},{"id":"2403.05534","title":"Bayesian Preference Elicitation with Language Models","zh_title":"基于语言模型的贝叶斯偏好诱导","primary_category":"cs.CL","date":"2024-03-08","score":5,"bucket":"other","tags":["偏好诱导","贝叶斯实验设计","人机交互"],"rubric_hits":["D1"],"abs_url":"https://arxiv.org/abs/2403.05534","has_summary":true},{"id":"2402.19421","title":"Crafting Knowledge: Exploring the Creative Mechanisms of Chat-Based Search Engines","zh_title":"知识构建：探索基于聊天的搜索引擎的创造性机制","primary_category":"cs.IR","date":"2024-02-29","score":0,"bucket":"other","tags":["搜索引擎","信息检索","语言模型偏好"],"rubric_hits":["C4"],"abs_url":"https://arxiv.org/abs/2402.19421","has_summary":false},{"id":"2402.18144","title":"Random Silicon Sampling: Simulating Human Sub-Population Opinion Using a Large Language Model Based on Group-Level Demographic Information","zh_title":"随机硅采样：基于群体人口统计信息用大语言模型模拟人类子群体意见","primary_category":"cs.AI","date":"2024-02-28","score":10,"bucket":"selected","tags":["LLM人类仿真","意见模拟","算法偏差"],"rubric_hits":["A1","A2","A5","B1","B4"],"abs_url":"https://arxiv.org/abs/2402.18144","has_summary":true},{"id":"2402.01053","title":"Plan-Grounded Large Language Models for Dual Goal Conversational Settings","zh_title":"基于计划的大语言模型用于双目标对话设置","primary_category":"cs.CL","date":"2024-02-01","score":0,"bucket":"other","tags":["对话系统","计划引导","混合主动"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2402.01053","has_summary":false},{"id":"2402.01766","title":"LLM Voting: Human Choices and AI Collective Decision Making","zh_title":"LLM投票：人类选择与AI集体决策","primary_category":"cs.CL","date":"2024-01-31","score":9,"bucket":"selected","tags":["LLM仿真","投票行为","人类对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2402.01766","has_summary":true},{"id":"2401.15589","title":"OpineBot: Class Feedback Reimagined Using a Conversational LLM","zh_title":"OpineBot：用对话式大语言模型重塑课堂反馈","primary_category":"cs.HC","date":"2024-01-28","score":0,"bucket":"other","tags":["课堂反馈","对话系统","人机交互"],"rubric_hits":["C3"],"abs_url":"https://arxiv.org/abs/2401.15589","has_summary":false},{"id":"2401.07345","title":"Can an LLM Learn Preferences from Choice Data?","zh_title":"大语言模型能从选择数据中学习偏好吗？","primary_category":"econ.GN","date":"2024-01-14","score":7,"bucket":"pending","tags":["偏好学习","选择实验","经济学决策"],"rubric_hits":["A1","B1","B2"],"abs_url":"https://arxiv.org/abs/2401.07345","has_summary":true},{"id":"2304.03442","title":"Generative Agents: Interactive Simulacra of Human Behavior","zh_title":"生成式智能体：人类行为的交互式模拟","primary_category":"cs.HC","date":"2023-04-07","score":8,"bucket":"selected","tags":["LLM社会模拟","人类行为仿真","智能体架构"],"rubric_hits":["A3","B1"],"abs_url":"https://arxiv.org/abs/2304.03442","has_summary":true},{"id":"2301.07543","title":"Large Language Models as Simulated Economic Agents: What Can We Learn from Homo Silicus?","zh_title":"作为模拟经济主体的大语言模型：我们能从Homo Silicus中学到什么？","primary_category":"econ.GN","date":"2023-01-18","score":10,"bucket":"selected","tags":["LLM仿真","经济实验","人类行为对照"],"rubric_hits":["A1","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2301.07543","has_summary":true},{"id":"2209.06899","title":"Out of One, Many: Using Language Models to Simulate Human Samples","zh_title":"一生万物：使用语言模型模拟人类样本","primary_category":"cs.LG","date":"2022-09-14","score":10,"bucket":"selected","tags":["LLM仿真","算法保真度","社会调查"],"rubric_hits":["A1","A2","A3","B1","B2"],"abs_url":"https://arxiv.org/abs/2209.06899","has_summary":true},{"id":"2208.10264","title":"Using Large Language Models to Simulate Multiple Humans and Replicate Human Subject Studies","zh_title":"使用大语言模型模拟多个人类并复现人类被试研究","primary_category":"cs.CL","date":"2022-08-18","score":10,"bucket":"selected","tags":["LLM仿真","人类实验复现","行为经济学"],"rubric_hits":["A1","A2","A3","B1","B2","B4"],"abs_url":"https://arxiv.org/abs/2208.10264","has_summary":true}]}