{
  "claimSet": {
    "claims": [
      {
        "boundary": "结论主要来自特定任务（报童、翻牌前RFI）和三个LLM家族，且人类基准数据较旧，向其他高风险场景的迁移有待验证。",
        "claimType": "cross_paper_synthesis",
        "contradictingPaperIds": [],
        "id": "C1",
        "maturity": "supported",
        "statement": "LLM在扑克和报童等决策任务中复制并放大了人类系统性偏差，例如GPT-4在报童任务中平均订购偏差比人类基准高70%，在扑克翻牌前决策中偏离GTO策略；这些偏差在提供最优公式后仍部分存在，表明其根源在于模型架构而非知识缺口。",
        "supportingPaperIds": [
          "arxiv:2512.12552",
          "arxiv:2308.12466"
        ],
        "whyItMatters": "这意味着在部署AI辅助决策时必须假定其会犯人类式错误，且更先进的模型不一定更可靠，需要额外校准和监督。"
      },
      {
        "boundary": "证据来自扑克和Liar's Dice等博弈任务，且PokerSkill可能对GTOWizard过拟合，规则化提示在其他领域的效果需要验证。",
        "claimType": "cross_paper_synthesis",
        "contradictingPaperIds": [],
        "id": "C2",
        "maturity": "supported",
        "statement": "在LLM辅助的不完全信息决策中，显式提供领域规则或求解器蒸馏策略能显著提升决策质量，例如SCCS将LLM与求解器策略的平均L1距离从0.211降至0.100，PokerSkill将GPT-5.5在HUNL中的损失从-132降至-57 mbb/hand，说明规则约束比模型自由推理更接近理性基准。",
        "supportingPaperIds": [
          "openalex:W7202230800",
          "openalex:W7162893802"
        ],
        "whyItMatters": "这为现实应用提供了低成本缓解AI决策偏差的路径：将专家知识编码为可执行规则，而非追求更大模型或更多训练。"
      },
      {
        "boundary": "证据来自简化扑克环境（Leduc、Kuhn）和综述性讨论，在完整德州扑克或非平稳对手下的适用性未知。",
        "claimType": "cross_paper_synthesis",
        "contradictingPaperIds": [],
        "id": "C3",
        "maturity": "supported",
        "statement": "在复杂不完全信息博弈中，精确GTO策略因计算不可行而无法直接应用，而针对次优对手的剥削性策略可以在保持对纳什均衡对手稳健性的同时获得更高收益，例如AlphaExploitem在Leduc Hold'em上对次优对手平均约+1.0 BBs/hand，而GTO基准收益低于此。",
        "supportingPaperIds": [
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "whyItMatters": "这提示现实风险管理不能只追求不可剥削的防御性策略，还需要识别和利用对手或市场的系统性偏差，但需权衡模型风险。"
      },
      {
        "boundary": "结论基于LLM生成的生态任务和已发表行为数据，且未包含工作记忆等约束，生态先验可能存在分布偏差。",
        "claimType": "direct_finding",
        "contradictingPaperIds": [],
        "id": "C4",
        "maturity": "single_source",
        "statement": "人类在函数学习、类别学习和决策制定中的行为可以通过元学习生态先验的模型（ERMI）优于经典认知模型来解释，例如在函数学习插值中MSE为0.0171，低于经典模型的0.0256，表明许多认知偏差可能是对环境统计结构的理性适应而非纯粹缺陷。",
        "supportingPaperIds": [
          "arxiv:2509.00116"
        ],
        "whyItMatters": "这改变对偏差的负面定性，提示设计决策支持系统时应考虑环境结构，而不是简单纠正表面行为。"
      },
      {
        "boundary": "该论文无实证，且'认知主权'未操作化，扑克与董事会情境在反馈结构和对手性质上可能差异很大。",
        "claimType": "evidence_boundary",
        "contradictingPaperIds": [],
        "id": "C5",
        "maturity": "single_source",
        "statement": "扑克训练被理论化为能够培育领导者的'认知主权'（包括概率推理、情绪调节和战略欺骗），但这一主张目前仅基于理论类比，缺乏任何实证数据，因此不能作为扑克技能迁移到现实领导决策的直接证据。",
        "supportingPaperIds": [
          "openalex:W4411621406"
        ],
        "whyItMatters": "提醒读者对'扑克智慧可迁移到商业'的流行说法保持警惕，需要实验证据而非仅靠类比。"
      },
      {
        "boundary": "理论保证仅限两人零和博弈，且使用固定下注大小抽象，扩展到多人或更复杂环境尚不可行。",
        "claimType": "direct_finding",
        "contradictingPaperIds": [],
        "id": "C6",
        "maturity": "single_source",
        "statement": "基于强化学习与搜索的算法（如ReBeL）在不完全信息博弈中能够达到超人水平，例如在双人无限注德州扑克中以165±69千分之一bb/手击败人类专家Dong Kim，但其训练需要128台机器每台8 GPU，表明完全理性算法在现实高风险决策中的直接应用受限于计算成本。",
        "supportingPaperIds": [
          "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        ],
        "whyItMatters": "这为理解人类与AI决策差异提供基准，也提示资源受限场景下需要近似或启发式方法。"
      }
    ],
    "evidenceMode": "fulltext",
    "schemaVersion": 1
  },
  "editorialPlan": {
    "centralThesis": "围绕“德州扑克策略的现实启示：决策心理学、风险管理与人类认知：关于德州扑克策略与技巧的研究结论，对于理解人类决策、风险管理和认知偏差有哪些启示？这些研究如何在现实场景中应用或改变我们的思维？”，应先区分当前证据直接支持的结论与仍待验证的推断。",
    "evidenceMode": "fulltext",
    "modules": [
      {
        "argumentRole": "orient",
        "avoidRepeatingClaimIds": [],
        "claimIds": [],
        "confidencePolicy": "evidence_calibrated",
        "confusionToResolve": "",
        "evidencePaperIds": [
          "openalex:W4411621406",
          "arxiv:2308.12466",
          "arxiv:2509.23747"
        ],
        "exampleRequirement": "none",
        "id": "M1",
        "includeReason": "先建立读者理解后续结论所需的共同语境。",
        "kind": "orientation",
        "lengthBudget": 300,
        "readerQuestion": "这项研究问题的范围和阅读入口是什么？",
        "readerTakeaway": "先明确问题范围和阅读入口。",
        "renderMode": "prose",
        "requirements": [
          "说明问题边界",
          "避免把研究背景写成结论"
        ],
        "title": "如何理解这个问题",
        "transitionFromPrevious": "开篇建立共同语境。"
      },
      {
        "argumentRole": "answer",
        "avoidRepeatingClaimIds": [],
        "claimIds": [
          "C1",
          "C2",
          "C3",
          "C4",
          "C5",
          "C6"
        ],
        "confidencePolicy": "evidence_calibrated",
        "confusionToResolve": "",
        "evidencePaperIds": [
          "openalex:W4411621406",
          "arxiv:2308.12466",
          "arxiv:2509.23747",
          "arxiv:2509.00116",
          "arxiv:2512.12552",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "exampleRequirement": "concrete_example",
        "id": "M2",
        "includeReason": "让读者先获得能够独立理解的结论。",
        "kind": "core_conclusions",
        "lengthBudget": 1100,
        "readerQuestion": "当前证据最直接支持哪些结论？",
        "readerTakeaway": "读者能够复述当前证据支持的核心认识。",
        "renderMode": "prose",
        "requirements": [
          "每条结论说明重要性",
          "结论与证据强度相匹配"
        ],
        "title": "目前可以带走的核心结论",
        "transitionFromPrevious": "在问题定向后直接回答研究问题。"
      },
      {
        "argumentRole": "synthesize",
        "avoidRepeatingClaimIds": [],
        "claimIds": [
          "C1",
          "C2",
          "C3",
          "C4",
          "C5",
          "C6"
        ],
        "confidencePolicy": "evidence_calibrated",
        "confusionToResolve": "",
        "evidencePaperIds": [
          "openalex:W4411621406",
          "arxiv:2308.12466",
          "arxiv:2509.23747",
          "arxiv:2509.00116",
          "arxiv:2512.12552",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "exampleRequirement": "contrast_pair",
        "id": "M3",
        "includeReason": "帮助读者理解不同工作之间的关系。",
        "kind": "research_landscape",
        "lengthBudget": 900,
        "readerQuestion": "现有研究主要从哪些问题入口展开？",
        "readerTakeaway": "读者能够理解不同工作围绕哪些问题形成分支。",
        "renderMode": "map",
        "requirements": [
          "按问题而不是论文顺序组织",
          "说明各分支之间的关系"
        ],
        "title": "当前研究版图",
        "transitionFromPrevious": "核心结论之后解释这些认识在研究版图中的关系。"
      },
      {
        "argumentRole": "assess_evidence",
        "avoidRepeatingClaimIds": [],
        "claimIds": [
          "C1",
          "C2",
          "C3",
          "C4",
          "C5",
          "C6"
        ],
        "confidencePolicy": "evidence_calibrated",
        "confusionToResolve": "",
        "evidencePaperIds": [
          "openalex:W4411621406",
          "arxiv:2308.12466",
          "arxiv:2509.23747",
          "arxiv:2509.00116",
          "arxiv:2512.12552",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "exampleRequirement": "none",
        "id": "M4",
        "includeReason": "防止把有限证据写成领域共识。",
        "kind": "evidence_boundaries",
        "lengthBudget": 700,
        "readerQuestion": "当前证据没有回答什么？",
        "readerTakeaway": "读者能够区分已获支持的判断与仍待验证的推断。",
        "renderMode": "prose",
        "requirements": [
          "区分缺失信息和反对证据",
          "明确仍需全文或新研究验证的部分"
        ],
        "title": "这些结论能相信到什么程度",
        "transitionFromPrevious": "在全文结尾校准前述判断的适用范围。"
      }
    ],
    "narrativeArc": [
      "先明确问题范围和阅读入口。",
      "读者能够复述当前证据支持的核心认识。",
      "读者能够理解不同工作围绕哪些问题形成分支。",
      "读者能够区分已获支持的判断与仍待验证的推断。"
    ],
    "omittedModules": [
      {
        "kind": "method_evolution",
        "reason": "未在规划阶段确认足够清晰的问题—方法—代价演进链。"
      },
      {
        "kind": "system_layers",
        "reason": "未在规划阶段确认稳定的系统层级关系。"
      }
    ],
    "readerTakeaways": [
      "LLM在扑克和报童等决策任务中复制并放大了人类系统性偏差，例如GPT-4在报童任务中平均订购偏差比人类基准高70%，在扑克翻牌前决策中偏离GTO策略；这些偏差在提供最优公式后仍部分存在，表明其根源在于模型架构而非知识缺口。",
      "在LLM辅助的不完全信息决策中，显式提供领域规则或求解器蒸馏策略能显著提升决策质量，例如SCCS将LLM与求解器策略的平均L1距离从0.211降至0.100，PokerSkill将GPT-5.5在HUNL中的损失从-132降至-57 mbb/hand，说明规则约束比模型自由推理更接近理性基准。",
      "在复杂不完全信息博弈中，精确GTO策略因计算不可行而无法直接应用，而针对次优对手的剥削性策略可以在保持对纳什均衡对手稳健性的同时获得更高收益，例如AlphaExploitem在Leduc Hold'em上对次优对手平均约+1.0 BBs/hand，而GTO基准收益低于此。",
      "人类在函数学习、类别学习和决策制定中的行为可以通过元学习生态先验的模型（ERMI）优于经典认知模型来解释，例如在函数学习插值中MSE为0.0171，低于经典模型的0.0256，表明许多认知偏差可能是对环境统计结构的理性适应而非纯粹缺陷。",
      "扑克训练被理论化为能够培育领导者的'认知主权'（包括概率推理、情绪调节和战略欺骗），但这一主张目前仅基于理论类比，缺乏任何实证数据，因此不能作为扑克技能迁移到现实领导决策的直接证据。",
      "基于强化学习与搜索的算法（如ReBeL）在不完全信息博弈中能够达到超人水平，例如在双人无限注德州扑克中以165±69千分之一bb/手击败人类专家Dong Kim，但其训练需要128台机器每台8 GPU，表明完全理性算法在现实高风险决策中的直接应用受限于计算成本。"
    ],
    "schemaVersion": 2
  },
  "evidenceCards": {
    "cards": [
      {
        "baseline": {
          "detail": "",
          "status": "not_applicable"
        },
        "claim": "这篇理论论文主张扑克训练能培育领导者的认知主权，但它没有提供任何实证数据，并且将扑克与董事会情境直接类比可能过度简化，因此不能作为扑克技能迁移到现实领导决策的直接证据。",
        "cost": {
          "detail": "",
          "status": "not_applicable"
        },
        "direct_evidence": false,
        "evidence_level": "theory_position",
        "evidence_role": "counter",
        "failure_conditions": [
          "作者未定义或测量“认知主权”，无法判断扑克玩家是否真的具有该特征。",
          "该论文将扑克策略与领导力进行类比，但二者在反馈结构、对手性质和决策后果上可能差异很大。",
          "论文未说明情绪调节和训练方案的具体实施细节，无法验证其主张。"
        ],
        "id": "EC001",
        "independent_context": "领导力与战略决策中的心理韧性，无原始数据，仅基于文献综述和理论整合。",
        "limitations": [
          "无原始实证数据，结论完全依赖文献推论。",
          "未提供扑克玩家与对照组的任何行为测量。",
          "“认知主权”缺乏可操作化定义。"
        ],
        "paper_id": "openalex:W4411621406",
        "quantitative_result": false,
        "question_id": "RQ4",
        "result": {
          "detail": "",
          "status": "not_reported"
        },
        "source_basis": "fulltext",
        "support_direction": "contradicts",
        "task": {
          "detail": "",
          "status": "not_applicable"
        },
        "validation": {
          "detail": "",
          "status": "not_applicable"
        }
      },
      {
        "baseline": {
          "detail": "",
          "status": "not_applicable"
        },
        "claim": "该文从理论层面提出，扑克专家通过训练获得概率推理、情绪调节和战略欺骗能力，而新手在这些方面可能不足，但这一专家优势的差异缺乏行为数据验证。",
        "cost": {
          "detail": "",
          "status": "not_applicable"
        },
        "direct_evidence": false,
        "evidence_level": "theory_position",
        "evidence_role": "synthesis",
        "failure_conditions": [
          "未提供专家与新手的行为对比数据，无法确认优势存在。",
          "训练过程未说明具体方法和时长，难以推断专家优势来源。"
        ],
        "id": "EC002",
        "independent_context": "领导力与战略决策中的认知训练，文献综述，无样本。",
        "limitations": [
          "论文无实证，所有结论为理论推论。",
          "未定义专家与新手的操作化标准。"
        ],
        "paper_id": "openalex:W4411621406",
        "quantitative_result": false,
        "question_id": "RQ2",
        "result": {
          "detail": "",
          "status": "not_reported"
        },
        "source_basis": "fulltext",
        "support_direction": "qualifies",
        "task": {
          "detail": "",
          "status": "not_applicable"
        },
        "validation": {
          "detail": "",
          "status": "not_applicable"
        }
      },
      {
        "baseline": {
          "detail": "与GTO翻牌前RFI参考图表（Little）比较。",
          "status": "reported"
        },
        "claim": "ChatGPT和GPT-4在翻牌前RFI决策中都偏离GTO，ChatGPT过于保守而GPT-4过于激进，且GPT-4在按钮位按要求采用GTO时会加注90%手牌，说明LLM实际行为与理性GTO基准存在明显差距。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "仅覆盖翻牌前RFI这一最简单的决策点，无法推断后续回合。",
          "确定性采样未考虑对手建模，可能高估实际游戏能力。",
          "查询成本高但未报告，限制可重复性。"
        ],
        "id": "EC003",
        "independent_context": "9人桌无限注德州扑克翻牌前加注首入场景，LLM决策。",
        "limitations": [
          "仅分析翻牌前RFI，后续回合未评估。",
          "未考虑完整对手建模与多人互动。",
          "查询成本未报告。"
        ],
        "paper_id": "arxiv:2308.12466",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "ChatGPT频繁弃牌并出现limp，GPT-4从不limp且加注范围过宽；两种模型均非GTO玩家。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "qualifies",
        "task": {
          "detail": "在9人桌无限注德州扑克翻牌前加注首入（RFI）场景中，对169种起手牌×8个位置进行决策。",
          "status": "reported"
        },
        "validation": {
          "detail": "通过固定采样参数（温度0.2，top-p 0.95）和多次查询取多数决策与GTO图表比对。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "与元学习推理（MI）及多种领域特定模型（如RMC、GCM、PM、Rule）比较。",
          "status": "reported"
        },
        "claim": "通过LLM生成生态有效任务并用元学习训练得到的ERMI模型，在函数学习、类别学习和决策制定中逐试验预测人类行为均优于经典认知模型，表明人类决策可能是对环境统计结构的理性适应。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "LLM生成任务可能存在分布偏差，影响生态先验的准确性。",
          "未考虑工作记忆、注意力等认知约束，可能遗漏系统性偏差。",
          "部分类别学习任务类型上与人类行为有差异。"
        ],
        "id": "EC004",
        "independent_context": "函数学习、类别学习和决策制定任务，使用LLM生成任务和已发表的人类行为数据。",
        "limitations": [
          "训练数据由LLM生成，存在潜在偏差。",
          "模型忽略认知约束，可能不完整。",
          "特征选择可能不全面。"
        ],
        "paper_id": "arxiv:2509.00116",
        "quantitative_result": true,
        "question_id": "RQ1",
        "result": {
          "detail": "ERMI在函数学习插值中MSE=0.0171（MI=0.0256），在Shepard类别学习MSE=0.03（MI=0.26），在决策制定后验模型频率最高。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "supports",
        "task": {
          "detail": "函数学习、类别学习和决策制定任务，分别使用约1万个LLM生成任务，并利用已发表的人类行为数据。",
          "status": "reported"
        },
        "validation": {
          "detail": "在15个实验上逐试验预测，并使用最大似然估计、BIC和后验模型频率进行模型比较。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "与人类决策者基准（Schweitzer and Cachon 2000）比较；并比较有无最优公式提示两种条件。",
          "status": "reported"
        },
        "claim": "在动态报童决策中，GPT-4、GPT-4o和LLaMA-8B复制并放大了人类订购偏差，且提供最优公式后偏差仍存在，说明LLM的决策偏差源于架构约束而非知识缺口，挑战了理性计算模型的预期。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "模拟报童问题可能不完全代表现实供应链复杂性。",
          "人类基准数据来自2000年，可能过时。",
          "仅测试三个LLM，结论普适性有限。"
        ],
        "id": "EC005",
        "independent_context": "动态报童库存决策，使用三种LLM并对比人类基准。",
        "limitations": [
          "实验环境为模拟报童问题。",
          "人类基准数据较旧。",
          "未覆盖更多LLM架构。"
        ],
        "paper_id": "arxiv:2512.12552",
        "quantitative_result": true,
        "question_id": "RQ1",
        "result": {
          "detail": "无公式提示下GPT-4在低利润均匀分布偏差+100.25，比人类基准+59.06高70%；风险中性环境中偏差仍存在；提供公式后GPT-4o收敛斜率接近零，GPT-4出现+1.191。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "qualifies",
        "task": {
          "detail": "动态报童库存决策，15轮，三种需求分布（均匀、正态、对数正态）。",
          "status": "reported"
        },
        "validation": {
          "detail": "每条件10次独立重复，使用温度1.0，API及本地部署。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "无公式提示的默认条件作为对照。",
          "status": "reported"
        },
        "claim": "动态报童实验显示，提供结构化最优公式或规则化提示可以显著改善LLM的决策轨迹，而依赖模型自主推理可能放大偏差，因此在高风险决策中应引入规则约束和人机协同监督来管理AI建议风险。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "deployment",
        "failure_conditions": [
          "该结论基于报童库存任务，向金融投资等领域的迁移需要验证。",
          "仅覆盖三个模型，无法保证所有AI系统适用。"
        ],
        "id": "EC006",
        "independent_context": "运营管理中的AI辅助决策，基于报童问题实验。",
        "limitations": [
          "实验环境与真实金融决策存在差异。",
          "未测试规则化提示在其他风险决策任务中的效果。"
        ],
        "paper_id": "arxiv:2512.12552",
        "quantitative_result": true,
        "question_id": "RQ5",
        "result": {
          "detail": "GPT-4o在提供公式后收敛斜率接近零；GPT-4仍出现正收敛斜率，表明公式不能保证复杂模型最优。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "supports",
        "task": {
          "detail": "动态报童库存决策，评估三种LLM在有无最优公式提示下的表现。",
          "status": "reported"
        },
        "validation": {
          "detail": "每条件10次重复，定性分析模型输出。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "",
          "status": "not_applicable"
        },
        "claim": "精确GTO策略在限注德州扑克中因游戏树有3.6×10^17种变化而不可计算，实践者必须采用抽象、离散化下注和剥削性策略来近似或超越GTO，这体现了完全理性策略的计算不可行。",
        "cost": {
          "detail": "",
          "status": "not_applicable"
        },
        "direct_evidence": false,
        "evidence_level": "review_synthesis",
        "evidence_role": "synthesis",
        "failure_conditions": [
          "综述未提供多数策略的定量比较数据，只能作为定性背景。",
          "未涵盖最新深度学习机器人如Pluribus的详细实现。"
        ],
        "id": "EC007",
        "independent_context": "文献综述涵盖限注和无限注德州扑克，不涉及原始实验数据。",
        "limitations": [
          "缺少定量比较。",
          "主要结论基于早期系统和理论。"
        ],
        "paper_id": "arxiv:2401.06168",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "CFR+在HULHE中的可剥削性为0.986毫大盲每局；Loki等系统通过对手建模实现剥削性玩法。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "qualifies",
        "task": {
          "detail": "综述GTO扑克与剥削性扑克，涵盖限注和无限制德州扑克。",
          "status": "reported"
        },
        "validation": {
          "detail": "",
          "status": "not_applicable"
        }
      },
      {
        "baseline": {
          "detail": "与直接提示、原始摘要、Route-only（仅路径）以及Deep CFR比较。",
          "status": "reported"
        },
        "claim": "将求解器混合策略蒸馏为可解释规则后，8种LLM配置对求解器策略的L1距离从0.211降至0.100，相对改善52.6%，说明用规则化提示比直接给原始求解器输出更能弥合LLM与GTO之间的差距。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "L1距离是局部保真度，不能完全代表全局可利用性。",
          "硬稀疏化导致保真度损失。",
          "评估样本仅覆盖特定配置，可能不推广。"
        ],
        "id": "EC008",
        "independent_context": "两人NLH翻牌后和Liar's Dice决策，使用求解器输出训练LLM。",
        "limitations": [
          "硬稀疏化损失保真度。",
          "L1距离不等同可利用性。",
          "Liar's Dice中SCCS增量收益较小。"
        ],
        "paper_id": "openalex:W7202230800",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "SCCS使平均L1距离从0.211降至0.100；argmax一致率从57.2%升至76.1%；河牌端局MDT可剥削性低于Deep CFR。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "supports",
        "task": {
          "detail": "两人NLH翻牌后和Liar's Dice决策，使用超过2.5亿个求解器标记样本。",
          "status": "reported"
        },
        "validation": {
          "detail": "在未见目标手牌和Liar's Dice上验证策略可移植性。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "BabyTartanian8、Slumbot、LBR、Libratus等作为对照。",
          "status": "reported"
        },
        "claim": "ReBeL通过RL+搜索在双人无限注德州扑克中达到超人水平，以165±69千分之一bb/手击败人类专家Dong Kim，且使用的领域知识少于先前扑克AI，说明理性算法可以超越人类但需要大量计算资源。",
        "cost": {
          "detail": "训练数据生成使用128台机器、每台8 GPU。",
          "status": "reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "benchmark",
        "failure_conditions": [
          "输入维度随信息状态数量线性增长，在公共知识少的游戏中不可行。",
          "理论保证仅限两人零和博弈，无法推广到多玩家。",
          "计算成本高，资源受限场景难以复现。"
        ],
        "id": "EC009",
        "independent_context": "双人无限注德州扑克对局，与人类专家和扑克AI比较。",
        "limitations": [
          "CFR-AVG修改版理论健全性未证明。",
          "扑克测试使用固定下注大小抽象。",
          "人类对局结果可能有统计不确定性。"
        ],
        "paper_id": "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "ReBeL击败BabyTartanian8和Slumbot；以165±69千分之一bb/手击败Dong Kim；在TEH中250次迭代利用性约0.01。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "supports",
        "task": {
          "detail": "双人无限注德州扑克（HUNL）对局，7500手。",
          "status": "reported"
        },
        "validation": {
          "detail": "使用AIVAT方差削减，并给出两人零和博弈的理论收敛保证。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "与默认提示的GPT-5.5、Claude Opus 4.6/4.7以及Slumbot比较。",
          "status": "reported"
        },
        "claim": "PokerSkill将人类专家规则库与LLM推理结合，在HUNL中使GPT-5.5损失从-132降至-57 mbb/hand，且所有代理均优于Slumbot，说明结构化领域知识可以显著提升LLM在不完全信息博弈中的决策。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "仅对GTOWizard基准评估，可能过拟合。",
          "Slumbot比较为间接，置信区间不可比。",
          "模型API版本更新可能改变结果。"
        ],
        "id": "EC010",
        "independent_context": "双人无限注德州扑克，LLM与人类专家规则库结合。",
        "limitations": [
          "可能对GTOWizard过拟合，未在真实人类对手上验证。",
          "未精确隔离规则引擎和LLM推理的贡献。"
        ],
        "paper_id": "openalex:W7162893802",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "PokerSkill将GPT-5.5损失从-132降至-57 mbb/hand，Claude Opus 4.6从-204至-80，Claude Opus 4.7从-170至-87，均优于Slumbot。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "supports",
        "task": {
          "detail": "双人无限注德州扑克（HUNL），每个实验至少5000手牌。",
          "status": "reported"
        },
        "validation": {
          "detail": "使用AIVAT方差削减后的mbb/hand评估，温度1.0。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "随机策略、CFR、DeepCFR、MCCFR、NFSP。",
          "status": "reported"
        },
        "claim": "在合成NLHE决策状态中，不同CFR变体的GTO收敛指标随玩家数变化，CFR在3人局Top-1最高为0.478，而MCCFR在6人局Top-1与随机相同，说明博弈中的玩家数量会调节策略学习和均衡收敛。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "direct_primary",
        "failure_conditions": [
          "合成状态假设无噪声且对手手牌独立，可能无法代表真实牌局。",
          "算法仅在理想两人零和游戏中有收敛保证，多人局结果可能不稳。",
          "固定超参数和有限种子，结果可能变化。"
        ],
        "id": "EC011",
        "independent_context": "合成NLHE决策状态，单挑、3人局和6人局评估。",
        "limitations": [
          "未验证剥削对手策略的实际盈利性。",
          "扩展到完整多人NLHE需要更多资源。"
        ],
        "paper_id": "arxiv:2509.23747",
        "quantitative_result": true,
        "question_id": "RQ2",
        "result": {
          "detail": "MCCFR单挑Top-1=1.000、KL=0.015、CE=0.891；3人局CFR Top-1=0.478；6人局MCCFR CE最低但Top-1与随机相同。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "qualifies",
        "task": {
          "detail": "合成NLHE决策状态，单挑、3人局和6人局评估。",
          "status": "reported"
        },
        "validation": {
          "detail": "使用Top-1、KL散度、交叉熵及多人NashConv启发式评价。",
          "status": "reported"
        }
      },
      {
        "baseline": {
          "detail": "与AlphaHoldem基线及掩蔽上下文消融比较，并对照纳什均衡策略。",
          "status": "reported"
        },
        "claim": "AlphaExploitem在Leduc Hold'em上对次优对手获得约+1.0 BBs/hand，同时保持接近纳什均衡的稳健性，说明严格的GTO策略并非在所有对手下都是收益最高的选择，存在通过利用对手偏差提升收益的边界。",
        "cost": {
          "detail": "",
          "status": "not_reported"
        },
        "direct_evidence": true,
        "evidence_level": "primary_experiment",
        "evidence_role": "counter",
        "failure_conditions": [
          "仅评估平稳对手，未涵盖非平稳策略。",
          "Leduc复杂度低于完整德州扑克。",
          "训练数据效率低，计算成本高。"
        ],
        "id": "EC012",
        "independent_context": "Kuhn Poker和Leduc Hold'em环境，使用PGX库。",
        "limitations": [
          "玩具环境可能与真实扑克有差距。",
          "有限时间视野无法处理无限历史。"
        ],
        "paper_id": "openalex:W7161090269",
        "quantitative_result": true,
        "question_id": "RQ3",
        "result": {
          "detail": "Leduc上AlphaExploitem ID奖励约+1.0 BBs/hand vs AlphaHoldem +0.5；掩蔽后降至+0.54；对NE策略无明显退化。",
          "status": "reported"
        },
        "source_basis": "fulltext",
        "support_direction": "contradicts",
        "task": {
          "detail": "Kuhn Poker和Leduc Hold'em环境，使用PGX库。",
          "status": "reported"
        },
        "validation": {
          "detail": "8个随机种子。",
          "status": "reported"
        }
      }
    ],
    "coverage": [
      {
        "baseline_paper_ids": [
          "arxiv:2509.00116",
          "arxiv:2512.12552"
        ],
        "counter_paper_ids": [
          "arxiv:2512.12552"
        ],
        "direct_paper_ids": [
          "arxiv:2509.00116",
          "arxiv:2512.12552"
        ],
        "evidence_card_ids": [
          "EC004",
          "EC005"
        ],
        "independent_context_count": 2,
        "missing_requirements": [
          "direct_studies:2/3"
        ],
        "quantitative_paper_ids": [
          "arxiv:2509.00116",
          "arxiv:2512.12552"
        ],
        "question": "德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
        "question_id": "RQ1",
        "status": "partial"
      },
      {
        "baseline_paper_ids": [
          "arxiv:2509.23747"
        ],
        "counter_paper_ids": [
          "openalex:W4411621406",
          "arxiv:2509.23747"
        ],
        "direct_paper_ids": [
          "arxiv:2509.23747"
        ],
        "evidence_card_ids": [
          "EC002",
          "EC011"
        ],
        "independent_context_count": 2,
        "missing_requirements": [
          "direct_studies:1/2"
        ],
        "quantitative_paper_ids": [
          "arxiv:2509.23747"
        ],
        "question": "扑克专家与新手之间的决策差异受哪些环境条件（如经验、反馈、压力）影响？这些条件如何在不同扑克情境（如现金局与锦标赛）中调节决策能力？",
        "question_id": "RQ2",
        "status": "partial"
      },
      {
        "baseline_paper_ids": [
          "arxiv:2308.12466",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "openalex:W7161090269"
        ],
        "counter_paper_ids": [
          "arxiv:2308.12466",
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "direct_paper_ids": [
          "arxiv:2308.12466",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "openalex:W7161090269"
        ],
        "evidence_card_ids": [
          "EC003",
          "EC007",
          "EC008",
          "EC009",
          "EC010",
          "EC012"
        ],
        "independent_context_count": 6,
        "missing_requirements": [],
        "quantitative_paper_ids": [
          "arxiv:2308.12466",
          "arxiv:2401.06168",
          "openalex:W7202230800",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "openalex:W7162893802",
          "openalex:W7161090269"
        ],
        "question": "在扑克研究中，理性博弈论最优策略与人类实际行为之间的差距体现了哪些权衡？这一差距对设计现实决策辅助工具（如AI建议系统）有何启示？",
        "question_id": "RQ3",
        "status": "answered"
      },
      {
        "baseline_paper_ids": [],
        "counter_paper_ids": [
          "openalex:W4411621406"
        ],
        "direct_paper_ids": [],
        "evidence_card_ids": [
          "EC001"
        ],
        "independent_context_count": 1,
        "missing_requirements": [
          "direct_studies:0/2",
          "independent_contexts:1/2"
        ],
        "quantitative_paper_ids": [],
        "question": "扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？扑克研究对决策心理学理论（如生态理性）的贡献与争议如何界定其结论的适用边界？",
        "question_id": "RQ4",
        "status": "partial"
      },
      {
        "baseline_paper_ids": [
          "arxiv:2512.12552"
        ],
        "counter_paper_ids": [],
        "direct_paper_ids": [
          "arxiv:2512.12552"
        ],
        "evidence_card_ids": [
          "EC006"
        ],
        "independent_context_count": 1,
        "missing_requirements": [
          "direct_studies:1/3",
          "independent_contexts:1/2"
        ],
        "quantitative_paper_ids": [
          "arxiv:2512.12552"
        ],
        "question": "扑克玩家的风险管理策略（如资金管理、止损规则）在金融投资等现实高风险决策中的有效性证据如何？在直觉与理性计算之间，扑克研究如何指导实践者的下一步选择？",
        "question_id": "RQ5",
        "status": "partial"
      }
    ],
    "output_language": "zh-CN",
    "paper_ids": [
      "openalex:W4411621406",
      "arxiv:2308.12466",
      "arxiv:2509.23747",
      "arxiv:2509.00116",
      "arxiv:2512.12552",
      "openalex:W7202230800",
      "s2:df2b23787a58b10962951d4f663809eb20a828ea",
      "openalex:W7162893802",
      "arxiv:2401.06168",
      "openalex:W7161090269"
    ],
    "review_contract_input_fingerprint": "c5692d43faa85ec5727957867e6b316f458d415e013d8a319343f9b92087ac67",
    "schema_version": 2,
    "source_scope": "fulltext",
    "status": "ready",
    "summary": {
      "answered_question_count": 1,
      "card_count": 12,
      "direct_primary_paper_count": 8,
      "partial_question_count": 4,
      "unanswered_question_count": 0
    }
  },
  "insights": [
    {
      "boundary": "现有研究测量了报童库存和翻牌前 RFI 等单步决策，涉及三个 LLM 家族；未测量多轮互动、真实人类对手或金融投资等高风险场景。",
      "claimId": "C1",
      "confidence": "supported",
      "evidencePaperIds": [
        "arxiv:2512.12552",
        "arxiv:2308.12466"
      ],
      "explanation": "在库存决策和扑克翻牌前两个完全不同的任务里，GPT-4 与 ChatGPT 都表现出稳定偏离最优策略的模式，且偏离方向相反——一个过于保守，另一个过于激进。报童实验中，GPT-4 的订购偏差比人类基准高 70%，即使拿到最优公式后偏差仍部分存在，说明问题不在知识缺口，而在模型架构本身。",
      "id": "I1",
      "implication": "部署 AI 辅助决策时不能默认其可靠，需要额外校准和监督，尤其要警惕复杂模型因过度分析而犯错。",
      "summary": "在库存决策和扑克翻牌前两个完全不同的任务里，GPT-4 与 ChatGPT 都表现出稳定偏离最优策略的模式，且偏离方向相反——一个过于保守，另一个过于激进。报童实验中，GPT-4 的订购偏差比人类基准高 70%，即使拿到最优公式后偏差仍部分存在，说明问题不在知识缺口，而在模型架构本身。",
      "title": "AI 决策复制并放大人类偏差"
    },
    {
      "boundary": "证据来自两人扑克和 Liar's Dice 等博弈任务，且 PokerSkill 可能对 GTOWizard 基准过拟合；未验证其他领域或真实人类对手。",
      "claimId": "C2",
      "confidence": "supported",
      "evidencePaperIds": [
        "openalex:W7202230800",
        "openalex:W7162893802"
      ],
      "explanation": "两篇独立研究分别用求解器蒸馏规则和人类专家规则库约束 LLM 决策：前者将 LLM 与求解器策略的平均 L1 距离从 0.211 降到 0.100，后者将 GPT-5.5 在双人无限注德州扑克中的损失从 -132 改善到 -57 mbb/hand。两者共同表明，显式规则约束比让模型自己推理更有效。",
      "id": "I2",
      "implication": "在现实应用中应优先把专家知识编码为可执行规则或约束，而不是单纯追求更大模型或更多训练数据。",
      "summary": "两篇独立研究分别用求解器蒸馏规则和人类专家规则库约束 LLM 决策：前者将 LLM 与求解器策略的平均 L1 距离从 0.211 降到 0.100，后者将 GPT-5.5 在双人无限注德州扑克中的损失从 -132 改善到 -57 mbb/hand。两者共同表明，显式规则约束比让模型自己推理更有效。",
      "title": "规则约束比自由推理更接近理性"
    },
    {
      "boundary": "证据来自简化扑克环境（Leduc、Kuhn）和综述性讨论，未在完整德州扑克或非平稳对手下验证。",
      "claimId": "C3",
      "confidence": "supported",
      "evidencePaperIds": [
        "arxiv:2401.06168",
        "openalex:W7161090269"
      ],
      "explanation": "AlphaExploitem 在 Leduc Hold'em 上利用对手的次优策略获得约 +1.0 BBs/hand，而严格 GTO 基线收益更低；综述也指出人类对手会无意偏离 GTO，剥削性策略可提高期望收益。这改变了“GTO 永远最优”的抽象认识，但需要识别对手偏差并承担模型风险。",
      "id": "I3",
      "implication": "风险管理不能只追求不可剥削的防御性策略，还应主动识别并利用对手或市场的系统性偏差，同时控制误判成本。",
      "summary": "AlphaExploitem 在 Leduc Hold'em 上利用对手的次优策略获得约 +1.0 BBs/hand，而严格 GTO 基线收益更低；综述也指出人类对手会无意偏离 GTO，剥削性策略可提高期望收益。这改变了“GTO 永远最优”的抽象认识，但需要识别对手偏差并承担模型风险。",
      "title": "打最优牌不等于赚最多钱"
    },
    {
      "boundary": "结论基于 LLM 生成任务和已发表行为数据，未包含工作记忆等认知约束，且仅一篇研究，未在其他独立样本复制。",
      "claimId": "C4",
      "confidence": "single_source",
      "evidencePaperIds": [
        "arxiv:2509.00116"
      ],
      "explanation": "ERMI 模型通过元学习生态先验，在函数学习和类别学习任务上预测人类行为优于经典认知模型，例如函数学习插值 MSE 为 0.0171，低于经典模型的 0.0256。这与 C1 中 LLM 偏差源于架构缺陷形成对比，但本研究未直接比较人类与 LLM 的偏差来源。",
      "id": "I4",
      "implication": "设计决策支持系统时应考虑环境统计结构，把偏差视为环境适应的结果，而不是简单纠正表面行为。",
      "summary": "ERMI 模型通过元学习生态先验，在函数学习和类别学习任务上预测人类行为优于经典认知模型，例如函数学习插值 MSE 为 0.0171，低于经典模型的 0.0256。这与 C1 中 LLM 偏差源于架构缺陷形成对比，但本研究未直接比较人类与 LLM 的偏差来源。",
      "title": "认知偏差可能是适应而非缺陷"
    }
  ],
  "language": "zh-CN",
  "narrativeSections": [
    {
      "blocks": [
        {
          "claimIds": [
            "C1",
            "C2",
            "C3",
            "C4",
            "C5",
            "C6"
          ],
          "epistemicStatus": "editorial_inference",
          "evidencePaperIds": [
            "arxiv:2512.12552",
            "openalex:W7202230800",
            "openalex:W7162893802",
            "arxiv:2401.06168",
            "openalex:W7161090269",
            "arxiv:2509.00116",
            "s2:df2b23787a58b10962951d4f663809eb20a828ea",
            "openalex:W4411621406"
          ],
          "id": "S1-B1",
          "role": "answer",
          "text": "现有研究主要从三个问题入口展开：一是能否在计算资源有限时逼近不可剥削的均衡，又能否利用对手偏差超越均衡（对应主张 C3 与 C6）；二是 LLM 在报童和扑克等不确定任务中是否复制人类偏差，以及结构化规则能否纠正这些偏差（C1 与 C2）；三是把扑克当作认知实验室，追问人类偏差是否是对环境统计的适应，或扑克训练能否迁移到一般决策能力（C4 与 C5）。"
        },
        {
          "claimIds": [
            "C2",
            "C6"
          ],
          "epistemicStatus": "cross_paper_synthesis",
          "evidencePaperIds": [
            "s2:df2b23787a58b10962951d4f663809eb20a828ea",
            "openalex:W7162893802",
            "openalex:W7202230800"
          ],
          "id": "S1-B2",
          "role": "comparison",
          "text": "两条技术路线形成鲜明对照：纯算法路线（如 ReBeL）以大量计算换取接近纳什均衡的策略，训练数据生成需要 128 台机器、每台 8 GPU（C6）；而规则化 LLM 路线（如 PokerSkill 和 SCCS）无需训练或求解器即可将 GPT-5.5 在 HUNL 中的损失从 -132 降至 -57 mbb/hand，或把 8 种 LLM 配置对求解器策略的平均 L1 距离从 0.211 降到 0.100（C2）。前者提供理性上限但成本高昂，后者贴近应用但依赖专家规则并可能对基准过拟合。"
        },
        {
          "claimIds": [
            "C3",
            "C4"
          ],
          "epistemicStatus": "cross_paper_synthesis",
          "evidencePaperIds": [
            "arxiv:2401.06168",
            "openalex:W7161090269",
            "arxiv:2509.00116"
          ],
          "id": "S1-B3",
          "role": "explanation",
          "text": "第三个问题入口与“偏差是否合理”相连。一方面，GTO 综述指出精确均衡因限注德州扑克游戏树有 3.6×10^17 个节点而不可计算，剥削性策略可能比严格 GTO 获得更高收益（C3）；另一方面，元学习生态先验模型显示，人类在函数学习等任务中的行为可由对环境统计结构的适应来解释，而非纯粹缺陷（C4）。这两个结论共同提示：现实决策需要结合对环境结构的估计和对对手偏差的识别，而不是盲目追求理论最优。"
        },
        {
          "claimIds": [
            "C5"
          ],
          "epistemicStatus": "editorial_inference",
          "evidencePaperIds": [
            "openalex:W4411621406"
          ],
          "id": "S1-B4",
          "role": "boundary",
          "text": "最后，扑克训练向领导力等现实场景迁移的入口仍是理论空档：一篇理论论文提出“认知主权”但没有任何实证数据（C5）。因此，当前版图中 AI 决策偏差与规则化缓解、均衡计算与剥削策略已有初步实验支撑，而扑克技能向人类现实决策的因果迁移尚无任何受控实验，这构成后续研究需要补足的具体缺口。"
        }
      ],
      "confidence": "supported",
      "evidencePaperIds": [
        "arxiv:2512.12552",
        "openalex:W7202230800",
        "openalex:W7162893802",
        "arxiv:2401.06168",
        "openalex:W7161090269",
        "arxiv:2509.00116",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "openalex:W4411621406"
      ],
      "id": "S1",
      "kind": "research_landscape",
      "moduleId": "M3",
      "readerQuestion": "现有研究主要从哪些问题入口展开？",
      "title": "当前研究版图"
    },
    {
      "blocks": [
        {
          "claimIds": [
            "C5"
          ],
          "epistemicStatus": "editorial_inference",
          "evidencePaperIds": [
            "openalex:W4411621406"
          ],
          "id": "S2-B1",
          "role": "answer",
          "text": "当前证据没有回答扑克技能是否以及如何迁移到现实领导、投资或谈判决策；所有关于“认知主权”的论述都来自一篇无实证的理论文章，只建立了类比，没有在真人群体中测评扑克训练前后的决策变化。因此，从扑克研究到现实应用的直接因果推断目前是缺失的，而不是被反对的。"
        },
        {
          "claimIds": [
            "C1",
            "C2"
          ],
          "epistemicStatus": "cross_paper_synthesis",
          "evidencePaperIds": [
            "arxiv:2512.12552",
            "arxiv:2308.12466",
            "openalex:W7202230800",
            "openalex:W7162893802"
          ],
          "id": "S2-B2",
          "role": "evidence",
          "text": "第二，关于LLM复制人类偏差的结论仅来自少数模型和特定任务：报童实验只测了GPT-4、GPT-4o和LLaMA-8B，人类基准是2000年的数据；扑克实验只覆盖翻牌前RFI，且未模拟对手建模。更重要的是，规则约束并非普遍有效：在报童实验中，提供最优公式后GPT-4仍出现正收敛斜率+1.191，而GPT-4o接近最优；这说明“提供规则就能修正偏差”的叙述在复杂模型上可能失效，但尚未在金融或医疗等高风险领域复现。当前证据只能支持LLM在这些任务中表现出偏差，不能支持所有AI辅助决策都会如此。"
        },
        {
          "claimIds": [
            "C3",
            "C6"
          ],
          "epistemicStatus": "cross_paper_synthesis",
          "evidencePaperIds": [
            "arxiv:2401.06168",
            "openalex:W7161090269",
            "s2:df2b23787a58b10962951d4f663809eb20a828ea"
          ],
          "id": "S2-B3",
          "role": "evidence",
          "text": "第三，关于剥削性策略优于GTO的证据全部来自Kuhn Poker和Leduc Hold'em这些低复杂度环境，对手是固定的、平稳的；AlphaExploitem在Leduc上对次优池约+1.0 BBs/hand，但训练一个种子要12小时，且未在完整德州扑克或会调整策略的真人对手上测试。同时，综述指出限注德州扑克游戏树有3.6×10^17种变化，精确GTO不可计算，但两者结合并不能推出在真实扑克中剥削性策略更好——因为真实对手会观察并反制，而简化环境中的泛化性未知。此外，完全理性算法如ReBeL虽达到超人水平，但其理论保证仅限两人零和博弈，训练需128台机器每台8 GPU，资源门槛使直接移植到组织决策不可行。"
        },
        {
          "claimIds": [
            "C4"
          ],
          "epistemicStatus": "paper_finding",
          "evidencePaperIds": [
            "arxiv:2509.00116"
          ],
          "id": "S2-B4",
          "role": "boundary",
          "text": "第四，关于人类偏差可能是生态理性适应的结论仅由单一来源ERMI支持：该研究用LLM生成任务训练元学习模型，在函数学习插值MSE=0.0171优于经典模型，但训练数据来自LLM可能存在分布偏差；模型没有包含工作记忆、注意力等约束，且部分类别学习任务中与人类行为不一致。因此，这一结论目前是单一来源的假设，没有被独立人类行为数据或非LLM生成生态任务检验，不能作为普遍认知原则。"
        }
      ],
      "confidence": "supported",
      "evidencePaperIds": [
        "openalex:W4411621406",
        "arxiv:2512.12552",
        "arxiv:2308.12466",
        "openalex:W7202230800",
        "openalex:W7162893802",
        "arxiv:2401.06168",
        "openalex:W7161090269",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "arxiv:2509.00116"
      ],
      "id": "S2",
      "kind": "evidence_boundaries",
      "moduleId": "M4",
      "readerQuestion": "当前证据没有回答什么？",
      "title": "这些结论能相信到什么程度"
    }
  ],
  "papers": [
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "count": 0,
        "source": "openalex"
      },
      "firstPublished": "2025-01-01",
      "id": "openalex:W4411621406",
      "identifiers": {
        "doi": "10.56975/jnrid.v3i6.701486",
        "openalex": "W4411621406"
      },
      "importance": 2,
      "importanceReason": "Explains expert poker psychology, integrating cognitive science and behavioral economics, which forms a baseline for human decision-making under risk and uncertainty relevant to AI comparisons.",
      "introduction": {
        "keyFindings": [
          "论文未报告原始实证数据；所有发现均为基于文献的理论推论，如损失厌恶比例约2:1、工作记忆约4个信息块20秒等。"
        ],
        "limitations": [
          "无实证数据支持，结论缺乏经验基础。"
        ],
        "method": "文献综述与理论整合，无原始实验或数据采集。",
        "overview": "为扑克心理学与领导力决策提供理论桥梁，但证据强度低，需实证检验，主要用于提出研究问题而非回答它们。"
      },
      "isCore": true,
      "noteHref": "papers/841d0fc27ca591638c9fff81ded1c17a6a48f0578ee783ac8a97c82de1813922.md",
      "notePath": "papers/841d0fc27ca591638c9fff81ded1c17a6a48f0578ee783ac8a97c82de1813922.md",
      "pdfHref": "papers/841d0fc27ca591638c9fff81.pdf",
      "role": "application",
      "sourceUrl": "https://doi.org/10.56975/jnrid.v3i6.701486",
      "tags": [
        "理论与立场"
      ],
      "title": "The Psychology of the Poker Player: Neuro-Cognitive Models from Poker Table to the Boardroom",
      "venue": "openalex"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "source": "arxiv",
        "unavailableReason": "可信来源未提供引用数"
      },
      "firstPublished": "2023-08-23",
      "id": "arxiv:2308.12466",
      "identifiers": {
        "arxiv": "2308.12466"
      },
      "importance": 2,
      "importanceReason": "Directly evaluates LLM poker performance, revealing strengths and biases in decision-making under incomplete information, crucial for assessing AI capabilities.",
      "introduction": {
        "keyFindings": [
          "ChatGPT和GPT-4在翻牌前RFI场景均偏离GTO策略：ChatGPT过于保守（频繁弃牌、出现limp），GPT-4过于激进（从不limp、加注范围过宽）。",
          "提示形式显著影响模型决策，如'A4s'与'4As'导致不同决策矩阵。"
        ],
        "limitations": [
          "仅分析翻牌前RFI，最简单决策点，无法推断后续回合。"
        ],
        "method": "通过OpenAI API对169种起手牌×8个位置进行查询，使用系统提示和用户提示，ChatGPT每组合10次、GPT-4每组合5次取多数决策，与GTO参考图表（Little）比较。",
        "overview": "实证揭示LLM在扑克决策中表现出类似人类风格的系统性偏差，为AI决策偏差研究提供基础案例。"
      },
      "isCore": true,
      "noteHref": "papers/02f03e5432f0c53b3af697fda5545ebdb6a72bf0e68ecaa556a859cadcff395c.md",
      "notePath": "papers/02f03e5432f0c53b3af697fda5545ebdb6a72bf0e68ecaa556a859cadcff395c.md",
      "pdfHref": "papers/02f03e5432f0c53b3af697fd.pdf",
      "role": "evaluation",
      "sourceUrl": "https://arxiv.org/abs/2308.12466v2",
      "tags": [
        "cs.CL"
      ],
      "title": "Are ChatGPT and GPT-4 Good Poker Players? -- A Pre-Flop Analysis",
      "venue": "arxiv"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "source": "arxiv",
        "unavailableReason": "可信来源未提供引用数"
      },
      "firstPublished": "2025-09-28",
      "id": "arxiv:2509.23747",
      "identifiers": {
        "arxiv": "2509.23747"
      },
      "importance": 2,
      "importanceReason": "Introduces a method to train poker agents that exceed GTO strategies by exploiting opponents, advancing beyond classic minimax solutions.",
      "introduction": {
        "keyFindings": [
          "在合成NLHE决策状态上，MCCFR在单挑中500次迭代后Top-1=1.000、KL=0.015、CE=0.891，最接近GTO代理。",
          "3人局中CFR的Top-1=0.478，高于其他模型；6人局MCCFR的CE最低但Top-1与随机相同。"
        ],
        "limitations": [
          "合成数据可能无法完全代表真实牌局。"
        ],
        "method": "生成合成NLHE状态（特征street, equity, texture，分布0.4,0.3,0.2,0.1），用CFR、MCCFR、DeepCFR、NFSP自博弈训练，与GTO代理比较。",
        "overview": "探讨超越GTO的利润最大化路径，强调GTO作为防御基线的重要性，但剥削部分仅概念性，为后续剥削研究提供框架。"
      },
      "isCore": true,
      "noteHref": "papers/4258b4ebe4f602b379543b3d034043ca24b5cc8c0199a97f14b7e17c3f5bb8c1.md",
      "notePath": "papers/4258b4ebe4f602b379543b3d034043ca24b5cc8c0199a97f14b7e17c3f5bb8c1.md",
      "pdfHref": "papers/4258b4ebe4f602b379543b3d.pdf",
      "role": "method",
      "sourceUrl": "https://arxiv.org/abs/2509.23747v1",
      "tags": [
        "cs.GT"
      ],
      "title": "Beyond Game Theory Optimal: Profit-Maximizing Poker Agents for No-Limit Holdem",
      "venue": "arxiv"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "source": "arxiv",
        "unavailableReason": "可信来源未提供引用数"
      },
      "firstPublished": "2025-08-28",
      "id": "arxiv:2509.00116",
      "identifiers": {
        "arxiv": "2509.00116"
      },
      "importance": 2,
      "importanceReason": "Provides ecologically rational analysis framework linking human cognition to environmental statistics, using LLMs to generate tasks and meta-learning to derive models.",
      "introduction": {
        "keyFindings": [
          "ERMI在函数学习插值中MSE=0.0171，显著低于MI的0.0256；在类别学习Shepard任务中MSE=0.03 vs 0.26。",
          "在决策制定任务中，ERMI的后验模型频率最高（双属性实验0.7299，四属性实验0.6284）。"
        ],
        "limitations": [
          "LLM生成任务可能存在分布偏差或噪声，未系统评估影响。"
        ],
        "method": "使用LLM（Claude-V2）生成认知任务，通过元学习训练ERMI（Transformer架构），在函数学习、类别学习、决策制定上逐试验预测人类行为，并与多种认知模型比较。",
        "overview": "提供生态理性分析框架解释人类决策偏差，支持'决策是适应环境结构'的观点，为理解扑克中的启发式与偏差提供理论背景。"
      },
      "isCore": true,
      "noteHref": "papers/fa7bb79f070920a7e012b4bfb54c5adb7cca25d29eaa45e7511c77a66b19066d.md",
      "notePath": "papers/fa7bb79f070920a7e012b4bfb54c5adb7cca25d29eaa45e7511c77a66b19066d.md",
      "pdfHref": "papers/fa7bb79f070920a7e012b4bf.pdf",
      "role": "foundation",
      "sourceUrl": "https://arxiv.org/abs/2509.00116v3",
      "tags": [
        "q-bio.NC"
      ],
      "title": "Meta-learning ecological priors from large language models explains human learning and decision making",
      "venue": "arxiv"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "source": "arxiv",
        "unavailableReason": "可信来源未提供引用数"
      },
      "firstPublished": "2025-12-14",
      "id": "arxiv:2512.12552",
      "identifiers": {
        "arxiv": "2512.12552"
      },
      "importance": 2,
      "importanceReason": "Evaluates LLMs on a classic decision-making problem, finding amplified cognitive biases and a 'paradox of intelligence', crucial for safe deployment in operations.",
      "introduction": {
        "keyFindings": [
          "无公式提示下，LLMs复制'过低/过高'订购偏差，GPT-4在低利润均匀分布下偏差+100.25，比人类基准+59.06高70%。",
          "风险中性环境中偏差仍存在（GPT-4均匀高利润-1.16%，LLaMA-8B-10.95%，GPT-4o≤0.11%），证明偏差非源于风险规避。"
        ],
        "limitations": [
          "模拟报童问题可能不反映现实供应链复杂性。"
        ],
        "method": "动态报童实验，15轮，三种需求分布（均匀、正态、对数正态），每条件10次重复，比较有无最优公式提示，使用温度1.0。",
        "overview": "实证展示LLM复制并放大人类认知偏差，揭示'智能悖论'，对理解AI辅助决策的风险和偏差管理有直接启示。"
      },
      "isCore": true,
      "noteHref": "papers/125945ef97465c7a1946f52cc8c5ca2b0133def6f8a95b36e6f037c4482123a5.md",
      "notePath": "papers/125945ef97465c7a1946f52cc8c5ca2b0133def6f8a95b36e6f037c4482123a5.md",
      "pdfHref": "papers/125945ef97465c7a1946f52c.pdf",
      "role": "evaluation",
      "sourceUrl": "https://arxiv.org/abs/2512.12552v1",
      "tags": [
        "cs.AI"
      ],
      "title": "Large Language Newsvendor: Decision Biases and Cognitive Mechanisms",
      "venue": "arxiv"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "count": 0,
        "source": "openalex"
      },
      "firstPublished": "2026-08-07",
      "id": "openalex:W7202230800",
      "identifiers": {
        "arxiv": "2608.06741",
        "openalex": "W7202230800",
        "semanticScholar": "d9c5a9da073a04ab23b5e40d30275cbc829a9890"
      },
      "importance": 2,
      "importanceReason": "Proposes Mixed-Strategy Decision Tree to distill solver output into interpretable strategic rules, improving LLM equilibrium reasoning without human data.",
      "introduction": {
        "keyFindings": [
          "SCCS规则将8种LLM配置的平均L1距离从0.211降至0.100，相对改善52.6%。",
          "argmax动作一致率从直接提示的57.2%提升至SCCS的76.1%。"
        ],
        "limitations": [
          "硬稀疏化导致保真度损失（硬MDT L1=0.087 vs 密集模型0.021）。"
        ],
        "method": "使用商业求解器生成250M+混合策略决策，训练混合策略决策树（MDT），用场景约束反事实采样（SCCS）提取对比规则，在NLH和Liar's Dice上评估LLM表现。",
        "overview": "展示如何将求解器知识转化为可理解规则来改善LLM决策，强调可解释性在决策辅助中的价值。"
      },
      "isCore": true,
      "noteHref": "papers/6b4beffe0a5c426115f5bb10f4e6e962afafee3ba1df585c3dd746ef74bb579a.md",
      "notePath": "papers/6b4beffe0a5c426115f5bb10f4e6e962afafee3ba1df585c3dd746ef74bb579a.md",
      "pdfHref": "papers/6b4beffe0a5c426115f5bb10.pdf",
      "role": "method",
      "sourceUrl": "https://arxiv.org/abs/2608.06741",
      "tags": [
        "方法与系统"
      ],
      "title": "Solver-Guided Reasoning for Mixed-Equilibrium Strategies",
      "venue": "openalex"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "count": 196,
        "source": "semantic_scholar"
      },
      "firstPublished": "2020-07-27",
      "id": "s2:df2b23787a58b10962951d4f663809eb20a828ea",
      "identifiers": {
        "arxiv": "2007.13544",
        "semanticScholar": "df2b23787a58b10962951d4f663809eb20a828ea"
      },
      "importance": 2,
      "importanceReason": "Introduces ReBeL, a general framework for self-play RL and search that achieves superhuman performance in heads-up NLH with minimal domain knowledge.",
      "introduction": {
        "keyFindings": [
          "ReBeL在TEH中250次迭代后利用性约0.01，相当于tabular CFR约125次迭代的水平。",
          "在HUNL中击败BabyTartanian8、Slumbot，并以165±69千分之一bb/手击败人类专家Dong Kim（7500手）。"
        ],
        "limitations": [
          "输入维度随信息状态数量线性增长，在公共信息少的游戏中不可行。"
        ],
        "method": "将不完全信息博弈转换为公共信念状态（PBS）的完美信息博弈，训练值/策略网络，在深度受限子博弈内运行CFR搜索，训练和测试时均使用搜索，测试时随机选择迭代保证安全。",
        "overview": "里程碑式工作，证明RL+搜索可在不完美信息博弈中达到超人水平，为AI在不确定环境下的决策提供方法原型。"
      },
      "isCore": true,
      "noteHref": "papers/2a7934967fa910d3b643b4baea4bca973b7f8fd090bea11bcaa00cbeda350b92.md",
      "notePath": "papers/2a7934967fa910d3b643b4baea4bca973b7f8fd090bea11bcaa00cbeda350b92.md",
      "pdfHref": "papers/2a7934967fa910d3b643b4ba.pdf",
      "role": "method",
      "sourceUrl": "https://www.semanticscholar.org/paper/df2b23787a58b10962951d4f663809eb20a828ea",
      "tags": [
        "理论与立场"
      ],
      "title": "Combining Deep Reinforcement Learning and Search for Imperfect-Information Games",
      "venue": "semantic_scholar"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "count": 0,
        "source": "openalex"
      },
      "firstPublished": "2026-05-28",
      "id": "openalex:W7162893802",
      "identifiers": {
        "arxiv": "2605.30094",
        "openalex": "W7162893802"
      },
      "importance": 3,
      "importanceReason": "作为核心精读论文，为研究问题提供直接证据。",
      "introduction": {
        "keyFindings": [
          "PokerSkill将GPT-5.5 XHigh损失从-132降至-57 mbb/hand（57%改善），Claude Opus 4.6从-204降至-80（61%），Claude Opus 4.7从-170降至-87（49%）。",
          "纯规则基线（无LLM）达到-132±19 mbb/hand，与默认提示LLM相当，但远低于PokerSkill+LLM的-57。"
        ],
        "limitations": [
          "Slumbot比较为间接，方差削减方法不同导致置信区间不可比。"
        ],
        "method": "构建由人类专家设计的分层技能库，上下文引擎检索相关技能并限制动作空间，ATT/DEF预算系统编码数值约束，LLM在受限选项中决策。",
        "overview": "展示LLM+人类规则的低成本路径达到专家级扑克，对AI辅助决策与人类专家知识结合具有启示。"
      },
      "isCore": true,
      "noteHref": "papers/da847dc26887e796f047cdd356f4014354b356682ceb58d320d08fc5c004ff05.md",
      "notePath": "papers/da847dc26887e796f047cdd356f4014354b356682ceb58d320d08fc5c004ff05.md",
      "pdfHref": "papers/da847dc26887e796f047cdd3.pdf",
      "role": "方法与系统",
      "sourceUrl": "https://arxiv.org/abs/2605.30094",
      "tags": [
        "方法与系统"
      ],
      "title": "PokerSkill: LLMs Can Play Expert-Level Poker without Training or Solvers",
      "venue": "openalex"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "source": "arxiv",
        "unavailableReason": "可信来源未提供引用数"
      },
      "firstPublished": "2024-01-02",
      "id": "arxiv:2401.06168",
      "identifiers": {
        "arxiv": "2401.06168"
      },
      "importance": 3,
      "importanceReason": "作为核心精读论文，为研究问题提供直接证据。",
      "introduction": {
        "keyFindings": [
          "限注德州扑克游戏树有3.6×10^17种变化，精确GTO计算不可行。",
          "人类对手会无意偏离GTO，可被剥削性策略利用以提高EV。"
        ],
        "limitations": [
          "缺乏定量比较数据。"
        ],
        "method": "文献综述，定性比较GTO与剥削策略、抽象技术、下注模型、多人游戏策略。",
        "overview": "提供GTO扑克的理论背景和挑战，奠定后续将扑克作为AI决策试验台的理解基础。"
      },
      "isCore": true,
      "noteHref": "papers/5045c8b84dd8f956252f5608bab8dd12b9ba9ca146a15a94d3940eddccc74837.md",
      "notePath": "papers/5045c8b84dd8f956252f5608bab8dd12b9ba9ca146a15a94d3940eddccc74837.md",
      "pdfHref": "papers/5045c8b84dd8f956252f5608.pdf",
      "role": "综述与文献回顾",
      "sourceUrl": "https://arxiv.org/abs/2401.06168v1",
      "tags": [
        "cs.GT"
      ],
      "title": "A Survey on Game Theory Optimal Poker",
      "venue": "arxiv"
    },
    {
      "citation": {
        "capturedAt": "2026-08-20",
        "count": 0,
        "source": "openalex"
      },
      "firstPublished": "2026-05-09",
      "id": "openalex:W7161090269",
      "identifiers": {
        "arxiv": "2605.09150",
        "openalex": "W7161090269"
      },
      "importance": 3,
      "importanceReason": "作为核心精读论文，为研究问题提供直接证据。",
      "introduction": {
        "keyFindings": [
          "在Leduc Hold'em上，AlphaExploitem对分布内玩具池奖励约+1.0 BBs/hand，AlphaHoldem约+0.5。",
          "掩蔽跨手牌上下文后，Leduc ID奖励从+1.10降至+0.54 BBs/hand，OOD从+1.16降至+0.52，约一半收益来自上下文。"
        ],
        "limitations": [
          "仅评估平稳对手，未考虑非平稳策略。"
        ],
        "method": "在AlphaHoldem架构上新增分层Transformer编码器处理过去手牌序列，训练中包含K-best联盟、固定弱策略池和动态长尾快照缓冲，在Kuhn和Leduc上评估。",
        "overview": "实证展示利用跨手牌历史信息超越纳什均衡，为适应性决策和对手建模提供启示。"
      },
      "isCore": true,
      "noteHref": "papers/2f7b8db2981556091661c8651e5675cc7bf5efbb1963a221bce66b88810d3d26.md",
      "notePath": "papers/2f7b8db2981556091661c8651e5675cc7bf5efbb1963a221bce66b88810d3d26.md",
      "pdfHref": "papers/2f7b8db2981556091661c865.pdf",
      "role": "方法与系统",
      "sourceUrl": "https://arxiv.org/abs/2605.09150",
      "tags": [
        "方法与系统"
      ],
      "title": "AlphaExploitem: Going Beyond the Nash Equilibrium in Poker by Learning to Exploit Suboptimal Play",
      "venue": "openalex"
    }
  ],
  "privacy": {
    "defaultPrivate": true,
    "forbiddenTerms": []
  },
  "readerQuestionCoverage": [
    {
      "baseline_paper_ids": [
        "arxiv:2509.00116",
        "arxiv:2512.12552"
      ],
      "counter_paper_ids": [
        "arxiv:2512.12552"
      ],
      "direct_paper_ids": [
        "arxiv:2509.00116",
        "arxiv:2512.12552"
      ],
      "evidence_card_ids": [
        "EC004",
        "EC005"
      ],
      "independent_context_count": 2,
      "missing_requirements": [
        "direct_studies:2/3"
      ],
      "quantitative_paper_ids": [
        "arxiv:2509.00116",
        "arxiv:2512.12552"
      ],
      "question": "德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
      "question_id": "RQ1",
      "status": "partial"
    },
    {
      "baseline_paper_ids": [
        "arxiv:2509.23747"
      ],
      "counter_paper_ids": [
        "openalex:W4411621406",
        "arxiv:2509.23747"
      ],
      "direct_paper_ids": [
        "arxiv:2509.23747"
      ],
      "evidence_card_ids": [
        "EC002",
        "EC011"
      ],
      "independent_context_count": 2,
      "missing_requirements": [
        "direct_studies:1/2"
      ],
      "quantitative_paper_ids": [
        "arxiv:2509.23747"
      ],
      "question": "扑克专家与新手之间的决策差异受哪些环境条件（如经验、反馈、压力）影响？这些条件如何在不同扑克情境（如现金局与锦标赛）中调节决策能力？",
      "question_id": "RQ2",
      "status": "partial"
    },
    {
      "baseline_paper_ids": [
        "arxiv:2308.12466",
        "openalex:W7202230800",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "openalex:W7162893802",
        "openalex:W7161090269"
      ],
      "counter_paper_ids": [
        "arxiv:2308.12466",
        "arxiv:2401.06168",
        "openalex:W7161090269"
      ],
      "direct_paper_ids": [
        "arxiv:2308.12466",
        "openalex:W7202230800",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "openalex:W7162893802",
        "openalex:W7161090269"
      ],
      "evidence_card_ids": [
        "EC003",
        "EC007",
        "EC008",
        "EC009",
        "EC010",
        "EC012"
      ],
      "independent_context_count": 6,
      "missing_requirements": [],
      "quantitative_paper_ids": [
        "arxiv:2308.12466",
        "arxiv:2401.06168",
        "openalex:W7202230800",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "openalex:W7162893802",
        "openalex:W7161090269"
      ],
      "question": "在扑克研究中，理性博弈论最优策略与人类实际行为之间的差距体现了哪些权衡？这一差距对设计现实决策辅助工具（如AI建议系统）有何启示？",
      "question_id": "RQ3",
      "status": "answered"
    },
    {
      "baseline_paper_ids": [],
      "counter_paper_ids": [
        "openalex:W4411621406"
      ],
      "direct_paper_ids": [],
      "evidence_card_ids": [
        "EC001"
      ],
      "independent_context_count": 1,
      "missing_requirements": [
        "direct_studies:0/2",
        "independent_contexts:1/2"
      ],
      "quantitative_paper_ids": [],
      "question": "扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？扑克研究对决策心理学理论（如生态理性）的贡献与争议如何界定其结论的适用边界？",
      "question_id": "RQ4",
      "status": "partial"
    },
    {
      "baseline_paper_ids": [
        "arxiv:2512.12552"
      ],
      "counter_paper_ids": [],
      "direct_paper_ids": [
        "arxiv:2512.12552"
      ],
      "evidence_card_ids": [
        "EC006"
      ],
      "independent_context_count": 1,
      "missing_requirements": [
        "direct_studies:1/3",
        "independent_contexts:1/2"
      ],
      "quantitative_paper_ids": [
        "arxiv:2512.12552"
      ],
      "question": "扑克玩家的风险管理策略（如资金管理、止损规则）在金融投资等现实高风险决策中的有效性证据如何？在直觉与理性计算之间，扑克研究如何指导实践者的下一步选择？",
      "question_id": "RQ5",
      "status": "partial"
    }
  ],
  "relatedReports": {
    "rapidBrief": {
      "evidenceMode": "abstract_only",
      "href": "rapid-brief.html",
      "label": "摘要级快速简报"
    }
  },
  "repositories": [],
  "researchIntelligence": {
    "citationGraph": {
      "corpusScope": "fulltext_plus_abstract_scan",
      "coverage": {
        "availablePaperCount": 19,
        "status": "available",
        "totalPaperCount": 40
      },
      "direction": "citing_to_referenced",
      "edges": [
        {
          "fromPaperId": "arxiv:2111.07295",
          "source": "semantic_scholar",
          "toPaperId": "s2:2f1085d977583f282d29e11ef8aef33307e36d07"
        },
        {
          "fromPaperId": "arxiv:2111.07295",
          "source": "verified_discovery_edge",
          "toPaperId": "s2:2f1085d977583f282d29e11ef8aef33307e36d07"
        },
        {
          "fromPaperId": "arxiv:2401.06168",
          "source": "semantic_scholar",
          "toPaperId": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9"
        },
        {
          "fromPaperId": "arxiv:2401.06168",
          "source": "verified_discovery_edge",
          "toPaperId": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9"
        },
        {
          "fromPaperId": "arxiv:2401.06168",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "arxiv:2401.06168",
          "source": "verified_discovery_edge",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "arxiv:2509.23747",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "openalex:W7161090269",
          "source": "semantic_scholar",
          "toPaperId": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9"
        },
        {
          "fromPaperId": "openalex:W7161090269",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "openalex:W7162893802",
          "source": "semantic_scholar",
          "toPaperId": "arxiv:2308.12466"
        },
        {
          "fromPaperId": "openalex:W7162893802",
          "source": "semantic_scholar",
          "toPaperId": "arxiv:2501.08328"
        },
        {
          "fromPaperId": "openalex:W7162893802",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "openalex:W7164973936",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "openalex:W7202006369",
          "source": "semantic_scholar",
          "toPaperId": "arxiv:2501.08328"
        },
        {
          "fromPaperId": "openalex:W7202006369",
          "source": "semantic_scholar",
          "toPaperId": "openalex:W7162893802"
        },
        {
          "fromPaperId": "openalex:W7202006369",
          "source": "semantic_scholar",
          "toPaperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf"
        },
        {
          "fromPaperId": "openalex:W7202006369",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "openalex:W7202230800",
          "source": "semantic_scholar",
          "toPaperId": "arxiv:2308.12466"
        },
        {
          "fromPaperId": "openalex:W7202230800",
          "source": "semantic_scholar",
          "toPaperId": "arxiv:2501.08328"
        },
        {
          "fromPaperId": "openalex:W7202230800",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
          "source": "semantic_scholar",
          "toPaperId": "openalex:W7159547552"
        },
        {
          "fromPaperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
          "source": "semantic_scholar",
          "toPaperId": "openalex:W7161090269"
        },
        {
          "fromPaperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
          "source": "semantic_scholar",
          "toPaperId": "openalex:W7162893802"
        },
        {
          "fromPaperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
          "source": "semantic_scholar",
          "toPaperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        },
        {
          "fromPaperId": "s2:fb7b874fecd6f6ac2218e949d3431b097d88a2b5",
          "source": "semantic_scholar",
          "toPaperId": "s2:580ef7ab46739661ac4153a882adbb8e0753f5d2"
        }
      ],
      "methodNote": "覆盖最终纳入的精读与摘要扫描论文，仅显示由精确论文标识确认的集合内引用；缺少边不等于论文之间没有关系。",
      "nodes": [
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "fulltext",
          "firstPublished": "2025-01-01",
          "foundation": 0.0,
          "frontier": 0.538704118075549,
          "importance": 2,
          "paperId": "openalex:W4411621406",
          "structuralRole": "frontier",
          "tags": [
            "理论与立场"
          ],
          "title": "The Psychology of the Poker Player: Neuro-Cognitive Models from Poker Table to the Boardroom"
        },
        {
          "bridge": 0.002206693637366679,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2023-08-23",
          "foundation": 0.04522847295329418,
          "frontier": 0.385,
          "importance": 2,
          "paperId": "arxiv:2308.12466",
          "structuralRole": "frontier",
          "tags": [
            "cs.CL"
          ],
          "title": "Are ChatGPT and GPT-4 Good Poker Players? -- A Pre-Flop Analysis"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2025-09-28",
          "foundation": 0.0,
          "frontier": 0.503272565207101,
          "importance": 2,
          "paperId": "arxiv:2509.23747",
          "structuralRole": "frontier",
          "tags": [
            "cs.GT"
          ],
          "title": "Beyond Game Theory Optimal: Profit-Maximizing Poker Agents for No-Limit Holdem"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "fulltext",
          "firstPublished": "2025-08-28",
          "foundation": 0.0,
          "frontier": 0.49500000000000005,
          "importance": 2,
          "paperId": "arxiv:2509.00116",
          "structuralRole": "frontier",
          "tags": [
            "q-bio.NC"
          ],
          "title": "Meta-learning ecological priors from large language models explains human learning and decision making"
        },
        {
          "bridge": 0.0,
          "community": 2,
          "evidenceLevel": "fulltext",
          "firstPublished": "2025-12-14",
          "foundation": 0.0,
          "frontier": 0.49500000000000005,
          "importance": 2,
          "paperId": "arxiv:2512.12552",
          "structuralRole": "frontier",
          "tags": [
            "cs.AI"
          ],
          "title": "Large Language Newsvendor: Decision Biases and Cognitive Mechanisms"
        },
        {
          "bridge": 0.1031261493196028,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2026-08-07",
          "foundation": 0.0,
          "frontier": 0.5946926636033918,
          "importance": 2,
          "paperId": "openalex:W7202230800",
          "structuralRole": "frontier",
          "tags": [
            "方法与系统"
          ],
          "title": "Solver-Guided Reasoning for Mixed-Equilibrium Strategies"
        },
        {
          "bridge": 0.23854358219933802,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2020-07-27",
          "foundation": 0.2908340306006431,
          "frontier": 0.3892346964408113,
          "importance": 2,
          "paperId": "s2:df2b23787a58b10962951d4f663809eb20a828ea",
          "structuralRole": "frontier",
          "tags": [
            "理论与立场"
          ],
          "title": "Combining Deep Reinforcement Learning and Search for Imperfect-Information Games"
        },
        {
          "bridge": 0.13276940051489516,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2026-05-28",
          "foundation": 0.03314476299033954,
          "frontier": 0.5946926636033918,
          "importance": 3,
          "paperId": "openalex:W7162893802",
          "structuralRole": "frontier",
          "tags": [
            "方法与系统"
          ],
          "title": "PokerSkill: LLMs Can Play Expert-Level Poker without Training or Solvers"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2024-01-02",
          "foundation": 0.0,
          "frontier": 0.470518741737598,
          "importance": 3,
          "paperId": "arxiv:2401.06168",
          "structuralRole": "frontier",
          "tags": [
            "cs.GT"
          ],
          "title": "A Survey on Game Theory Optimal Poker"
        },
        {
          "bridge": 0.002942258183155572,
          "community": 0,
          "evidenceLevel": "fulltext",
          "firstPublished": "2026-05-09",
          "foundation": 0.017870678662483413,
          "frontier": 0.58364046445728,
          "importance": 3,
          "paperId": "openalex:W7161090269",
          "structuralRole": "frontier",
          "tags": [
            "方法与系统"
          ],
          "title": "AlphaExploitem: Going Beyond the Nash Equilibrium in Poker by Learning to Exploit Suboptimal Play"
        },
        {
          "bridge": 0.038249356381022434,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2022-06-28",
          "foundation": 0.06870537698656023,
          "frontier": 0.4968542149273121,
          "importance": 1,
          "paperId": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9",
          "structuralRole": "frontier",
          "tags": [],
          "title": "AlphaHoldem: High-Performance Artificial Intelligence for Heads-Up No-Limit Poker via End-to-End Reinforcement Learning"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2024-11-02",
          "foundation": 0.0,
          "frontier": 0.44312172271968214,
          "importance": 1,
          "paperId": "openalex:W4404351335",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Preference-CFR$\\:$ Beyond Nash Equilibrium for Better Game Strategies"
        },
        {
          "bridge": 0.04656123574843693,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-08-06",
          "foundation": 0.0,
          "frontier": 0.8,
          "importance": 1,
          "paperId": "openalex:W7202006369",
          "structuralRole": "frontier",
          "tags": [],
          "title": "AV-AIVAT: 74x Cheaper Agent Evaluation with Certified Anytime-Valid Stopping in Imperfect-Information Games"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-05-31",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "arxiv:2606.01390",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Limit Continuous Poker: A Variant of Continuous Poker with Limited Bet Sizes"
        },
        {
          "bridge": 0.036925340198602434,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-30",
          "foundation": 0.015274084327856138,
          "frontier": 0.8091653098719107,
          "importance": 1,
          "paperId": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Agents That Certify Their Own Exploits: Confidence-Scheduled Restricted Responses for Safe Opponent Exploitation"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-04-28",
          "foundation": 0.017870678662483413,
          "frontier": 0.5531217227196821,
          "importance": 1,
          "paperId": "openalex:W7159547552",
          "structuralRole": "frontier",
          "tags": [],
          "title": "StratFormer: Adaptive Opponent Modeling and Exploitation in Imperfect-Information Games"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-02-01",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "openalex:W7202060719",
          "structuralRole": "frontier",
          "tags": [],
          "title": "On Creating Human Models in Poker with Deep Learning and Regularized Search"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-08-08",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "openalex:W7203644380",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Learning strategic poker decision-making with Large Language Models"
        },
        {
          "bridge": 0.1588819418904009,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2021-11-14",
          "foundation": 0.0,
          "frontier": 0.321825840795231,
          "importance": 1,
          "paperId": "arxiv:2111.07295",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Rational AI: A comparison of human and AI responses to triggers of economic irrationality in poker"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-06-11",
          "foundation": 0.0,
          "frontier": 0.558272565207101,
          "importance": 1,
          "paperId": "openalex:W7164973936",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Poker Arena: Multi-Axis Profiling of Strategic Reasoning and Memory in LLMs"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2024-08-28",
          "foundation": 0.0,
          "frontier": 0.5065809586904276,
          "importance": 1,
          "paperId": "openalex:W4402025026",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Play the Man, Not the Cards. (Homo)socialities and Masculine Positions in Poker"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-06-24",
          "foundation": 0.07637042163928066,
          "frontier": 0.6076608016342326,
          "importance": 1,
          "paperId": "s2:580ef7ab46739661ac4153a882adbb8e0753f5d2",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Variable Bound Tightening for Nash Equilibrium Computation in Multiplayer Imperfect-Information Games"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-06-29",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:4f3f90348d4117212e2c8a982327c1c61ea62cce",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Player Psychology and Decision-Making in Board Game COUP"
        },
        {
          "bridge": 0.0051489518205222505,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2025-01-14",
          "foundation": 0.0605025572811503,
          "frontier": 0.49500000000000005,
          "importance": 1,
          "paperId": "arxiv:2501.08328",
          "structuralRole": "frontier",
          "tags": [],
          "title": "PokerBench: Training Large Language Models to become Professional Poker Players"
        },
        {
          "bridge": 0.0,
          "community": 2,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-05-08",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "arxiv:2605.07789",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Analyzing Human Heuristics and Strategies in Everyday Decision-Making Conversations for Conversational AI Design"
        },
        {
          "bridge": 0.0,
          "community": 2,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-01-16",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "arxiv:2601.11049",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Predicting Biased Human Decision-Making with Large Language Models in Conversational Settings"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2019-02-01",
          "foundation": 0.03818521081964033,
          "frontier": 0.29424527211918994,
          "importance": 1,
          "paperId": "s2:2f1085d977583f282d29e11ef8aef33307e36d07",
          "structuralRole": "frontier",
          "tags": [],
          "title": "The role of affect in management decisions: A systematic review"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2020-11-09",
          "foundation": 0.0,
          "frontier": 0.22000000000000003,
          "importance": 1,
          "paperId": "arxiv:2011.04450",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Kuhn Poker with Cheating and Its Detection"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-06-28",
          "foundation": 0.0,
          "frontier": 0.593704118075549,
          "importance": 1,
          "paperId": "s2:fb7b874fecd6f6ac2218e949d3431b097d88a2b5",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Projected Exploitability Descent for Nash Equilibrium Computation in Multiplayer Imperfect-Information Games"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-08-07",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:c2388f1811e05ef63ae6ce0a862faa3365fda3b8",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Beyond the Black Box: Interpretable Models of Human Randomisation Failures"
        },
        {
          "bridge": 0.0,
          "community": 2,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2020-09-28",
          "foundation": 0.0,
          "frontier": 0.22000000000000003,
          "importance": 1,
          "paperId": "arxiv:2009.13368",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Using Resource-Rational Analysis to Understand Cognitive Biases in Interactive Data Visualizations"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-06",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:b7d662729e6a5dbc0cdc1f0fb4c08c04942abd11",
          "structuralRole": "frontier",
          "tags": [],
          "title": "QualGames: A Qualtrics implementation and a database of behavioral game theory tasks"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2020-05-30",
          "foundation": 0.0,
          "frontier": 0.22000000000000003,
          "importance": 1,
          "paperId": "arxiv:2006.02256",
          "structuralRole": "frontier",
          "tags": [],
          "title": "QuLBIT: Quantum-Like Bayesian Inference Technologies for Cognition and Decision"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-20",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:8442c9db779fdb40f1b4e3a58b10b295fe994bb4",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Using the game can't stop to inform the novel behavioural state of near-loss."
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-08-16",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:9730e4fe994e973176c6254f4edb62670f2a5ede",
          "structuralRole": "frontier",
          "tags": [],
          "title": "CoupVisor: Strategy Optimization by Round and Challenge Decision Support"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-09",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:c0809ef19fcdb420ab6272562680e3f9c9c12e42",
          "structuralRole": "frontier",
          "tags": [],
          "title": "From Rules to Nash Equilibria: A Lean 4 Case Study in Game-Theoretic Analysis of a Competitive Trading Card Game"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-25",
          "foundation": 0.0,
          "frontier": 0.55,
          "importance": 1,
          "paperId": "s2:a5aa6a63fc812f6c68c69e0e6ab11433a4da3f8e",
          "structuralRole": "frontier",
          "tags": [],
          "title": "On the Power of Deception in Repeated Games"
        },
        {
          "bridge": 0.0,
          "community": 2,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2020-10-15",
          "foundation": 0.0,
          "frontier": 0.22000000000000003,
          "importance": 1,
          "paperId": "arxiv:2010.07938",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Deciding Fast and Slow: The Role of Cognitive Biases in AI-assisted Decision-making"
        },
        {
          "bridge": 0.0,
          "community": 0,
          "evidenceLevel": "abstract_only",
          "firstPublished": "2026-07-07",
          "foundation": 0.0,
          "frontier": 0.5531217227196821,
          "importance": 1,
          "paperId": "s2:2becf3a004455e0a5ab9ec493ebf444c2edb8f64",
          "structuralRole": "frontier",
          "tags": [],
          "title": "A Gold-Standard Study of What Makes a Lightweight Game-Playing Agent Strong"
        },
        {
          "bridge": 0.0,
          "community": 1,
          "evidenceLevel": "abstract_only",
          "firstPublished": "",
          "foundation": 0.0,
          "frontier": 0.275,
          "importance": 1,
          "paperId": "s2:771c9e534f5ea1d2f406da7d197125cc2b43da04",
          "structuralRole": "frontier",
          "tags": [],
          "title": "Omission matters: Reframing omission as action reduces apparent loss aversion"
        }
      ]
    },
    "evidenceMatrix": {
      "claims": [
        {
          "contradictingPaperIds": [],
          "id": "C1",
          "maturity": "supported",
          "statement": "LLM在扑克和报童等决策任务中复制并放大了人类系统性偏差，例如GPT-4在报童任务中平均订购偏差比人类基准高70%，在扑克翻牌前决策中偏离GTO策略；这些偏差在提供最优公式后仍部分存在，表明其根源在于模型架构而非知识缺口。",
          "supportingPaperIds": [
            "arxiv:2512.12552",
            "arxiv:2308.12466"
          ]
        },
        {
          "contradictingPaperIds": [],
          "id": "C2",
          "maturity": "supported",
          "statement": "在LLM辅助的不完全信息决策中，显式提供领域规则或求解器蒸馏策略能显著提升决策质量，例如SCCS将LLM与求解器策略的平均L1距离从0.211降至0.100，PokerSkill将GPT-5.5在HUNL中的损失从-132降至-57 mbb/hand，说明规则约束比模型自由推理更接近理性基准。",
          "supportingPaperIds": [
            "openalex:W7202230800",
            "openalex:W7162893802"
          ]
        },
        {
          "contradictingPaperIds": [],
          "id": "C3",
          "maturity": "supported",
          "statement": "在复杂不完全信息博弈中，精确GTO策略因计算不可行而无法直接应用，而针对次优对手的剥削性策略可以在保持对纳什均衡对手稳健性的同时获得更高收益，例如AlphaExploitem在Leduc Hold'em上对次优对手平均约+1.0 BBs/hand，而GTO基准收益低于此。",
          "supportingPaperIds": [
            "arxiv:2401.06168",
            "openalex:W7161090269"
          ]
        },
        {
          "contradictingPaperIds": [],
          "id": "C4",
          "maturity": "single_source",
          "statement": "人类在函数学习、类别学习和决策制定中的行为可以通过元学习生态先验的模型（ERMI）优于经典认知模型来解释，例如在函数学习插值中MSE为0.0171，低于经典模型的0.0256，表明许多认知偏差可能是对环境统计结构的理性适应而非纯粹缺陷。",
          "supportingPaperIds": [
            "arxiv:2509.00116"
          ]
        },
        {
          "contradictingPaperIds": [],
          "id": "C5",
          "maturity": "single_source",
          "statement": "扑克训练被理论化为能够培育领导者的'认知主权'（包括概率推理、情绪调节和战略欺骗），但这一主张目前仅基于理论类比，缺乏任何实证数据，因此不能作为扑克技能迁移到现实领导决策的直接证据。",
          "supportingPaperIds": [
            "openalex:W4411621406"
          ]
        },
        {
          "contradictingPaperIds": [],
          "id": "C6",
          "maturity": "single_source",
          "statement": "基于强化学习与搜索的算法（如ReBeL）在不完全信息博弈中能够达到超人水平，例如在双人无限注德州扑克中以165±69千分之一bb/手击败人类专家Dong Kim，但其训练需要128台机器每台8 GPU，表明完全理性算法在现实高风险决策中的直接应用受限于计算成本。",
          "supportingPaperIds": [
            "s2:df2b23787a58b10962951d4f663809eb20a828ea"
          ]
        }
      ],
      "paperIds": [
        "openalex:W4411621406",
        "arxiv:2308.12466",
        "arxiv:2509.23747",
        "arxiv:2509.00116",
        "arxiv:2512.12552",
        "openalex:W7202230800",
        "s2:df2b23787a58b10962951d4f663809eb20a828ea",
        "openalex:W7162893802",
        "arxiv:2401.06168",
        "openalex:W7161090269"
      ]
    },
    "institutionLandscape": {
      "corpusScope": "fulltext_plus_abstract_scan",
      "coverage": {
        "availablePaperCount": 8,
        "status": "available",
        "totalPaperCount": 40
      },
      "institutions": [
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "CN",
          "fulltextPaperCount": 2,
          "id": "I99065089",
          "name": "Tsinghua University",
          "paperIds": [
            "openalex:W7202230800",
            "openalex:W7162893802",
            "openalex:W7202006369"
          ],
          "prominenceScore": 6,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 0,
          "countryCode": "CN",
          "fulltextPaperCount": 2,
          "id": "I4210116924",
          "name": "Chinese University of Hong Kong, Shenzhen",
          "paperIds": [
            "openalex:W7202230800",
            "openalex:W7162893802"
          ],
          "prominenceScore": 5,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 0,
          "countryCode": "GB",
          "fulltextPaperCount": 1,
          "id": "I126193024",
          "name": "London Metropolitan University",
          "paperIds": [
            "openalex:W4411621406"
          ],
          "prominenceScore": 2,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 0,
          "countryCode": "DE",
          "fulltextPaperCount": 1,
          "id": "I4210110470",
          "name": "Richard Wolf (Germany)",
          "paperIds": [
            "openalex:W7202230800"
          ],
          "prominenceScore": 2,
          "type": "company"
        },
        {
          "abstractOnlyPaperCount": 0,
          "countryCode": "CN",
          "fulltextPaperCount": 1,
          "id": "I183067930",
          "name": "Shanghai Jiao Tong University",
          "paperIds": [
            "openalex:W7202230800"
          ],
          "prominenceScore": 2,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 0,
          "countryCode": "CA",
          "fulltextPaperCount": 1,
          "id": "I4210127509",
          "name": "Vector Institute",
          "paperIds": [
            "openalex:W7202230800"
          ],
          "prominenceScore": 2,
          "type": "facility"
        },
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "IN",
          "fulltextPaperCount": 0,
          "id": "I154851008",
          "name": "Indian Institute of Technology Roorkee",
          "paperIds": [
            "openalex:W7164973936"
          ],
          "prominenceScore": 1,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "SE",
          "fulltextPaperCount": 0,
          "id": "I43968019",
          "name": "Karlstad University",
          "paperIds": [
            "openalex:W4402025026"
          ],
          "prominenceScore": 1,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "NL",
          "fulltextPaperCount": 0,
          "id": "I34352273",
          "name": "Maastricht University",
          "paperIds": [
            "openalex:W7159547552"
          ],
          "prominenceScore": 1,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "US",
          "fulltextPaperCount": 0,
          "id": "I63966007",
          "name": "Massachusetts Institute of Technology",
          "paperIds": [
            "openalex:W7202060719"
          ],
          "prominenceScore": 1,
          "type": "education"
        },
        {
          "abstractOnlyPaperCount": 1,
          "countryCode": "US",
          "fulltextPaperCount": 0,
          "id": "I143685665",
          "name": "Master's College",
          "paperIds": [
            "openalex:W7202060719"
          ],
          "prominenceScore": 1,
          "type": "education"
        }
      ],
      "label": "本论文集中的机构显著度",
      "methodNote": "覆盖最终纳入的精读与摘要扫描论文，按论文覆盖数和编辑重要性汇总；摘要论文仅计基础权重，不代表全球机构排名或机构质量评价。"
    },
    "researchCoordinates": [
      {
        "category": "理论与方法",
        "epistemicStatus": "cross_paper_synthesis",
        "evidencePaperIds": [
          "arxiv:2401.06168",
          "openalex:W7161090269"
        ],
        "name": "不完全信息博弈与纳什均衡",
        "whyItMatters": "理解博弈论最优策略（GTO）与剥削性策略的权衡是解读扑克决策结论的数学基础。"
      },
      {
        "category": "应用领域",
        "epistemicStatus": "paper_derived",
        "evidencePaperIds": [
          "arxiv:2512.12552",
          "arxiv:2308.12466"
        ],
        "name": "大语言模型决策偏差",
        "whyItMatters": "多个实验显示 LLM 在报童和扑克任务中复制并放大人类偏差，是评估 AI 辅助决策风险的关键证据。"
      },
      {
        "category": "系统与工程",
        "epistemicStatus": "cross_paper_synthesis",
        "evidencePaperIds": [
          "openalex:W7202230800",
          "openalex:W7162893802"
        ],
        "name": "求解器蒸馏与规则约束",
        "whyItMatters": "将求解器策略转化为可解释规则能显著提升 LLM 在扑克中的决策质量，为现实高风险场景提供低成本缓解路径。"
      },
      {
        "category": "理论与方法",
        "epistemicStatus": "paper_derived",
        "evidencePaperIds": [
          "arxiv:2509.00116"
        ],
        "name": "人类生态理性模型",
        "whyItMatters": "元学习生态先验模型为重新解释认知偏差提供了框架，挑战纯粹缺陷论。"
      }
    ],
    "schemaVersion": 1,
    "topicTerms": [
      {
        "category": "主题",
        "evidencePaperIds": [
          "arxiv:2512.12552",
          "arxiv:2308.12466"
        ],
        "term": "大语言模型决策偏差",
        "weight": 9
      },
      {
        "category": "方法",
        "evidencePaperIds": [
          "arxiv:2401.06168",
          "arxiv:2308.12466"
        ],
        "term": "博弈论最优策略（GTO）",
        "weight": 8
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "openalex:W7202230800",
          "openalex:W7162893802",
          "openalex:W7161090269"
        ],
        "term": "方法与系统",
        "weight": 8
      },
      {
        "category": "方法",
        "evidencePaperIds": [
          "openalex:W7202230800"
        ],
        "term": "求解器蒸馏规则",
        "weight": 8
      },
      {
        "category": "方法",
        "evidencePaperIds": [
          "openalex:W7161090269",
          "arxiv:2401.06168"
        ],
        "term": "剥削性策略",
        "weight": 7
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "arxiv:2509.23747",
          "arxiv:2401.06168"
        ],
        "term": "cs.GT",
        "weight": 6
      },
      {
        "category": "方法",
        "evidencePaperIds": [
          "arxiv:2509.00116"
        ],
        "term": "元学习生态先验",
        "weight": 6
      },
      {
        "category": "问题",
        "evidencePaperIds": [
          "arxiv:2512.12552"
        ],
        "term": "报童问题",
        "weight": 6
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "openalex:W4411621406",
          "s2:df2b23787a58b10962951d4f663809eb20a828ea"
        ],
        "term": "理论与立场",
        "weight": 6
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "arxiv:2512.12552"
        ],
        "term": "cs.AI",
        "weight": 4
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "arxiv:2308.12466"
        ],
        "term": "cs.CL",
        "weight": 4
      },
      {
        "category": "主题",
        "evidencePaperIds": [
          "arxiv:2509.00116"
        ],
        "term": "q-bio.NC",
        "weight": 4
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "决策心理学",
        "weight": 3
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "博弈论最优",
        "weight": 3
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "大语言模型",
        "weight": 3
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "德州扑克",
        "weight": 3
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "认知偏差",
        "weight": 3
      },
      {
        "category": "主题",
        "evidencePaperIds": [],
        "term": "风险管理",
        "weight": 3
      }
    ]
  },
  "researchStatus": {
    "assessment_retry_available": false,
    "assessment_round": 1,
    "automatic_supplement_round_limit": 1,
    "automatic_supplement_rounds_completed": 1,
    "completion_status": "completed_with_limitations",
    "continuation_available": true,
    "evidence_status": "insufficient",
    "final_high_priority_gap_count": 3,
    "initial_high_priority_gap_count": 3,
    "open_gaps": [
      {
        "evidence_paper_ids": [
          "openalex:W4411621406",
          "arxiv:2512.12552",
          "arxiv:2308.12466"
        ],
        "id": "GAP-RQ1-HUMAN",
        "missing_evidence": "人类玩家在真实德州扑克中的决策行为数据，如手牌选择、弃牌率、下注模式与期望值的关系，以及专家与新手在概率处理上的量化对比。",
        "queries": [
          "poker decision making expected value human players empirical study",
          "expert versus novice poker players probability judgment decisions"
        ],
        "question": "德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
        "rationale": "证据图谱中缺乏对人类扑克玩家的系统性实证研究，现有LLM偏差和理论综述无法提供人类决策的核心共识，若缺失将导致对期望值决策的结论缺乏直接人类证据，无法可靠挑战理性选择理论。",
        "reader_question_id": "RQ1",
        "severity": "high",
        "target_evidence": "primary"
      },
      {
        "evidence_paper_ids": [
          "arxiv:2509.00116",
          "openalex:W4411621406"
        ],
        "id": "GAP-RQ4-BOUNDARY",
        "missing_evidence": "关于扑克结论迁移到现实决策的边界条件研究，例如反馈频率、对抗性、情绪压力差异如何调节偏差表现，以及生态理性框架下的批评性实证。",
        "queries": [
          "boundary conditions transfer poker decision making to real world",
          "ecological rationality critique poker cognitive biases generality"
        ],
        "question": "扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？扑克研究对决策心理学理论（如生态理性）的贡献与争议如何界定其结论的适用边界？",
        "rationale": "证据图谱中缺少反驳或限制性证据来界定扑克结论的适用边界，尤其是与生态理性、自然决策的争议。没有这些边界证据，报告可能过度外推，误导现实应用。",
        "reader_question_id": "RQ4",
        "severity": "high",
        "target_evidence": "contradictory"
      },
      {
        "evidence_paper_ids": [
          "arxiv:2401.06168",
          "openalex:W7162893802"
        ],
        "id": "GAP-RQ5-FINANCE",
        "missing_evidence": "从扑克策略迁移到金融投资的风险管理实证研究，如资金管理规则在不同市场的适用性、止损规则对投资收益的影响，以及专家直觉在金融决策中的可靠性。",
        "queries": [
          "poker bankroll management principles applied to financial investment risk",
          "poker training transfer to financial decision making expert intuition"
        ],
        "question": "扑克玩家的风险管理策略（如资金管理、止损规则）在金融投资等现实高风险决策中的有效性证据如何？在直觉与理性计算之间，扑克研究如何指导实践者的下一步选择？",
        "rationale": "证据图谱中没有直接测试扑克风险管理策略（如资金管理、止损）在金融领域有效性的原始研究，也没有专家直觉可靠性在不同领域差异的证据。这会导致应用建议缺乏实证基础，无法回答研究问题的现实启示部分。",
        "reader_question_id": "RQ5",
        "severity": "high",
        "target_evidence": "primary"
      }
    ],
    "schema_version": 1,
    "stop_reason": "supplementary_reading_unavailable",
    "supplementary_completed_count": 0,
    "supplementary_selected_count": 1
  },
  "reviewContract": {
    "analysis_axis": [
      "决策机制与认知偏差",
      "风险管理与决策策略",
      "证据强度与迁移边界"
    ],
    "audience": "对决策心理学、行为经济学及现实领域（金融、商业）感兴趣的研究者、从业者与学习者",
    "audience_source": "system_inferred",
    "audit": {
      "checks": [
        "reader_value",
        "non_overlap",
        "evidence_answerability",
        "explicit_question_preservation"
      ],
      "status": "model_audited"
    },
    "candidate_questions": [
      {
        "answer_dimensions": [
          "扑克中期望值计算的实际运用与变异",
          "人类对概率信息的非理性处理方式",
          "理性选择理论在复杂博弈中的适用限度",
          "专家决策与常人决策的差异"
        ],
        "id": "CQ01",
        "question": "德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
        "role": "outcomes",
        "source_questions": [],
        "why_it_matters": "帮助读者把握扑克研究对决策科学的基础贡献，厘清理性模型与实证观察之间的关系"
      },
      {
        "answer_dimensions": [
          "情绪状态对风险偏好和判断准确性的影响",
          "前额叶功能与情绪调节的神经证据",
          "压力情境下认知资源的分配",
          "长期情绪管理与绩效的关联"
        ],
        "id": "CQ02",
        "question": "扑克中的“情绪失控”（tilt）通过哪些心理机制影响决策？研究证据如何支持或修正情绪调节模型？",
        "role": "mechanisms",
        "source_questions": [],
        "why_it_matters": "揭示情绪干扰决策的因果路径，为现实场景中的情绪管理提供理论依据"
      },
      {
        "answer_dimensions": [
          "资金管理规则与期望值优化的关系",
          "风险承受能力的个体差异",
          "止损规则在真实市场中的效果证据",
          "财政决策与扑克情境的结构相似性"
        ],
        "id": "CQ03",
        "question": "扑克玩家的风险管理策略（如资金管理、止损规则）是否具有跨情境有效性？它们在金融投资等现实场景中的适用条件是什么？",
        "role": "decisions",
        "source_questions": [],
        "why_it_matters": "有助于将扑克风险管理原则转化为其他高风险决策领域，同时识别适用前提"
      },
      {
        "answer_dimensions": [
          "前景理论中的损失厌恶在扑克中的表现",
          "过度自信对下注策略与胜负的影响",
          "禀赋效应、锚定效应在牌桌上的可复现性",
          "扑克现场研究与实验室研究的互校"
        ],
        "id": "CQ04",
        "question": "扑克研究提供了哪些证据来验证或质疑前景理论、过度自信等行为经济学预测？这些证据的可靠性如何？",
        "role": "evidence_quality",
        "source_questions": [],
        "why_it_matters": "评估扑克作为行为经济学实验室的实证价值，以及其结论对理论修正的作用"
      },
      {
        "answer_dimensions": [
          "扑克规则与日常决策结构的差异",
          "高频率反馈与低频率反馈对学习的影响",
          "对抗性环境与非对抗性环境的区分",
          "文化、个体差异的调节作用"
        ],
        "id": "CQ05",
        "question": "扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？",
        "role": "boundaries",
        "source_questions": [],
        "why_it_matters": "防止过度外推扑克结论，明确其生态效度边界"
      },
      {
        "answer_dimensions": [
          "延迟反馈与即时反馈对技能习得的影响",
          "训练强度与刻意练习的作用",
          "压力下专家决策的优势来源",
          "决策环境的结构化程度"
        ],
        "id": "CQ06",
        "question": "扑克专家与新手之间的决策差异反映了哪些环境条件（如经验、反馈、压力）的作用？这些条件如何塑造决策能力？",
        "role": "conditions",
        "source_questions": [],
        "why_it_matters": "揭示影响决策能力形成的关键情境因素，为培训和教育设计提供线索"
      },
      {
        "answer_dimensions": [
          "博弈论最优策略在真实扑克中的胜率优势",
          "人类计算能力与记忆限制对策略实施的影响",
          "对抗非理性对手时的策略调整",
          "辅助工具（如决策树）在现实中的适用性"
        ],
        "id": "CQ07",
        "question": "在扑克研究中，理性博弈论模型与人类实际行为之间的差距说明了什么权衡？这一差距对设计决策辅助工具有何启示？",
        "role": "tradeoffs",
        "source_questions": [],
        "why_it_matters": "帮助读者理解纯理性模型的局限性，平衡规范性与描述性决策理论"
      },
      {
        "answer_dimensions": [
          "概率思维训练对风险判断的提升",
          "情绪调节教学对冲动控制的影响",
          "培训效果的保持与迁移",
          "实施中遇到的障碍与失败因素"
        ],
        "id": "CQ08",
        "question": "针对非玩家的扑克决策培训（如概率思维训练）在改善现实决策方面的实证效果如何？有哪些成功或失败的实施案例？",
        "role": "implementation",
        "source_questions": [],
        "why_it_matters": "评估扑克启发的干预措施能否有效迁移至金融、管理等现实领域，指导实践应用"
      },
      {
        "answer_dimensions": [
          "扑克中沉没成本效应的操作化方式",
          "锚定效应在动态博弈中的测量",
          "生态化测量与实验室测量的信效度比较",
          "大数据与行为追踪在偏差识别中的应用"
        ],
        "id": "CQ09",
        "question": "扑克研究中对认知偏差（如沉没成本、锚定）的测量方法有哪些突破？这些方法能否为实验室外的决策研究提供工具？",
        "role": "mechanisms",
        "source_questions": [],
        "why_it_matters": "评估扑克情境作为认知偏差测量工具的优势与局限，促进研究方法的创新"
      },
      {
        "answer_dimensions": [
          "现金局与锦标赛的时间压力差异",
          "筹码深度对决策偏好的影响",
          "对手数量与博弈结构的变化",
          "短手牌与长手牌的认知负荷"
        ],
        "id": "CQ10",
        "question": "不同扑克情境中的决策研究发现是否一致？情境差异对结论的普遍性有什么影响？",
        "role": "conditions",
        "source_questions": [],
        "why_it_matters": "帮助读者理解研究结论的情境依赖性，避免忽略变体差异而错误外推"
      },
      {
        "answer_dimensions": [
          "专家直觉形成的模式识别机制",
          "直觉快速判断在时限压力下的准确性",
          "理性计算与直觉的交互作用",
          "如何训练可靠的直觉"
        ],
        "id": "CQ11",
        "question": "扑克研究如何解释直觉在决策中的作用？研究结果如何平衡直觉与理性计算在实际决策中的权重？",
        "role": "decisions",
        "source_questions": [],
        "why_it_matters": "回应现实决策中直觉与分析的撕扯，为决策风格选择提供实证依据"
      },
      {
        "answer_dimensions": [
          "生态理性在扑克决策中的体现",
          "自然决策与扑克快决策的异同",
          "扑克研究对启发式偏见的挑战",
          "理论整合与未解争论"
        ],
        "id": "CQ12",
        "question": "扑克研究对决策心理学理论（如生态理性、自然决策）的发展有何贡献？它与这些理论框架的主要争议点是什么？",
        "role": "boundaries",
        "source_questions": [],
        "why_it_matters": "定位扑克研究在决策科学理论谱系中的位置，理清其与其他框架的关系"
      }
    ],
    "explicit_user_questions": [],
    "input_fingerprint": "c5692d43faa85ec5727957867e6b316f458d415e013d8a319343f9b92087ac67",
    "question_count": 5,
    "questions": [
      {
        "acceptance_criteria": [
          "明确列出至少3个关于期望值决策的核心共识",
          "提供至少2项量化证据（如胜率差异、决策准确率）支持共识",
          "说明这些共识与传统理性选择理论的具体冲突点",
          "区分专家与新手在概率处理上的实证差异"
        ],
        "answer_dimensions": [
          "扑克中期望值计算的实际运用与变异",
          "人类对概率信息的非理性处理方式",
          "理性选择理论在复杂博弈中的适用限度",
          "专家决策与常人决策的差异"
        ],
        "id": "RQ1",
        "locked": false,
        "origin": "system_derived",
        "question": "德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
        "required_evidence": {
          "baseline_required": false,
          "counter_search_required": false,
          "minimum_direct_studies": 3,
          "minimum_independent_contexts": 2,
          "quantitative_result_required": true
        },
        "role": "outcomes",
        "search_facets": [
          "direct_primary",
          "synthesis",
          "frontier",
          "benchmark"
        ],
        "source_questions": [],
        "status": "unanswered",
        "why_it_matters": "帮助读者把握扑克研究对决策科学的基础贡献，厘清理性模型与实证观察之间的关系"
      },
      {
        "acceptance_criteria": [
          "识别至少3个影响专家与新手差异的环境条件",
          "比较不同扑克情境（如现金局与锦标赛）下决策表现的实证证据",
          "说明反馈、压力等条件如何塑造决策能力",
          "提供条件作用的调节机制解释"
        ],
        "answer_dimensions": [
          "延迟反馈与即时反馈对技能习得的影响",
          "训练强度与刻意练习的作用",
          "压力下专家决策的优势来源",
          "现金局与锦标赛的时间压力、筹码深度等结构差异",
          "对手数量与博弈结构对认知负荷的影响"
        ],
        "id": "RQ2",
        "locked": false,
        "origin": "system_derived",
        "question": "扑克专家与新手之间的决策差异受哪些环境条件（如经验、反馈、压力）影响？这些条件如何在不同扑克情境（如现金局与锦标赛）中调节决策能力？",
        "required_evidence": {
          "baseline_required": false,
          "counter_search_required": false,
          "minimum_direct_studies": 2,
          "minimum_independent_contexts": 2,
          "quantitative_result_required": false
        },
        "role": "conditions",
        "search_facets": [
          "direct_primary",
          "benchmark",
          "frontier"
        ],
        "source_questions": [],
        "status": "unanswered",
        "why_it_matters": "揭示影响决策能力形成的关键情境因素，为培训和教育设计提供线索，并澄清结论的适用条件"
      },
      {
        "acceptance_criteria": [
          "对比理性模型与人类表现的量化差异（如胜率、决策误差率）",
          "明确基准（如GTO策略或标准理性模型）",
          "分析至少2种权衡维度（如准确性vs.认知负荷、可解释性vs.复杂性）",
          "提出对现实决策辅助工具设计的可操作启示"
        ],
        "answer_dimensions": [
          "博弈论最优策略在真实扑克中的胜率优势",
          "人类计算能力与记忆限制对策略实施的影响",
          "对抗非理性对手时的策略调整",
          "辅助工具（如决策树、AI建议）在现实中的适用性",
          "采用理性策略的认知成本与收益权衡"
        ],
        "id": "RQ3",
        "locked": false,
        "origin": "system_derived",
        "question": "在扑克研究中，理性博弈论最优策略与人类实际行为之间的差距体现了哪些权衡？这一差距对设计现实决策辅助工具（如AI建议系统）有何启示？",
        "required_evidence": {
          "baseline_required": true,
          "counter_search_required": false,
          "minimum_direct_studies": 3,
          "minimum_independent_contexts": 2,
          "quantitative_result_required": true
        },
        "role": "tradeoffs",
        "search_facets": [
          "direct_primary",
          "benchmark",
          "counter",
          "frontier"
        ],
        "source_questions": [],
        "status": "unanswered",
        "why_it_matters": "帮助读者理解纯理性模型的局限性，平衡规范性与描述性决策理论，为工具设计提供依据"
      },
      {
        "acceptance_criteria": [
          "列出至少3个扑克结论不适用或需修正的现实场景",
          "提供反驳或限制性证据（counter）说明适用边界",
          "讨论扑克研究与生态理性、自然决策理论的关键争议",
          "说明情境变量（如反馈频率、对抗性）如何调节迁移效果"
        ],
        "answer_dimensions": [
          "扑克规则与日常决策结构的差异",
          "高频率反馈与低频率反馈对学习的影响",
          "对抗性环境与非对抗性环境的区分",
          "文化、个体差异的调节作用",
          "扑克研究对生态理性等理论的印证或挑战"
        ],
        "id": "RQ4",
        "locked": false,
        "origin": "system_derived",
        "question": "扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？扑克研究对决策心理学理论（如生态理性）的贡献与争议如何界定其结论的适用边界？",
        "required_evidence": {
          "baseline_required": false,
          "counter_search_required": true,
          "minimum_direct_studies": 2,
          "minimum_independent_contexts": 2,
          "quantitative_result_required": false
        },
        "role": "boundaries",
        "search_facets": [
          "direct_primary",
          "counter",
          "synthesis",
          "frontier"
        ],
        "source_questions": [],
        "status": "unanswered",
        "why_it_matters": "防止过度外推扑克结论，明确其生态效度边界及理论定位"
      },
      {
        "acceptance_criteria": [
          "区分扑克策略在金融等领域的适用与不适用条件",
          "提供至少2项资金管理或止损策略的实证效果数据",
          "说明直觉在专家决策中何时可靠、何时不可靠",
          "给出可操作的建议（如训练方法、决策框架调整）",
          "指出证据空白与未来研究方向"
        ],
        "answer_dimensions": [
          "资金管理规则与期望值优化的关系",
          "止损规则在真实市场中的效果证据",
          "专家直觉形成的模式识别机制及其在时限压力下的准确性",
          "理性计算与直觉的交互作用",
          "如何训练可靠的直觉并整合到决策流程"
        ],
        "id": "RQ5",
        "locked": false,
        "origin": "system_derived",
        "question": "扑克玩家的风险管理策略（如资金管理、止损规则）在金融投资等现实高风险决策中的有效性证据如何？在直觉与理性计算之间，扑克研究如何指导实践者的下一步选择？",
        "required_evidence": {
          "baseline_required": false,
          "counter_search_required": false,
          "minimum_direct_studies": 3,
          "minimum_independent_contexts": 2,
          "quantitative_result_required": false
        },
        "role": "decisions",
        "search_facets": [
          "direct_primary",
          "deployment",
          "benchmark",
          "synthesis"
        ],
        "source_questions": [],
        "status": "unanswered",
        "why_it_matters": "为实践者提供基于证据的策略选择建议，帮助其在直觉与分析之间取得平衡"
      }
    ],
    "reader_goal": "判断德州扑克研究对理解人类决策、风险管理和认知偏差的启示中，哪些具有可靠证据基础、可在现实场景中应用，以及应用的边界条件",
    "reader_goal_source": "system_inferred",
    "research_question": "德州扑克策略的现实启示：决策心理学、风险管理与人类认知：关于德州扑克策略与技巧的研究结论，对于理解人类决策、风险管理和认知偏差有哪些启示？这些研究如何在现实场景中应用或改变我们的思维？",
    "review_type": "broad_overview",
    "schema_version": 1,
    "status": "ready",
    "topic": "德州扑克策略的现实启示：决策心理学、风险管理与人类认知"
  },
  "schemaVersion": 3,
  "scope": {
    "boundaries": [
      "不细化到具体牌桌上的心理技巧，而侧重领域级启示",
      "不包含赌博成瘾的临床干预内容"
    ],
    "missingSources": [
      "semantic_scholar",
      "openalex",
      "补充检索后证据仍不足：德州扑克研究为理解人类在不确定条件下的期望值决策提供了哪些核心共识？这些共识如何挑战传统理性选择理论？",
      "补充检索后证据仍不足：扑克研究揭示的认知偏差结论向现实决策迁移时，存在哪些边界条件或不适用场景？扑克研究对决策心理学理论（如生态理性）的贡献与争议如何界定其结论的适用边界？",
      "补充检索后证据仍不足：扑克玩家的风险管理策略（如资金管理、止损规则）在金融投资等现实高风险决策中的有效性证据如何？在直觉与理性计算之间，扑克研究如何指导实践者的下一步选择？"
    ],
    "question": "德州扑克策略的现实启示：决策心理学、风险管理与人类认知：关于德州扑克策略与技巧的研究结论，对于理解人类决策、风险管理和认知偏差有哪些启示？这些研究如何在现实场景中应用或改变我们的思维？",
    "summary": "现有证据表明，德州扑克为研究不完全信息下的决策偏差和风险管理提供了可量化的试验台，并且通过求解器蒸馏或规则约束能显著改善 AI 辅助决策，但将扑克技能迁移到现实领导力的结论仍缺乏实证。",
    "topic": "德州扑克策略的现实启示：决策心理学、风险管理与人类认知"
  },
  "slug": "research-2932f5ae2f11",
  "subtitle": "用扑克和报童实验区分有证据支持的偏差结论与仅靠类比的领导力启示，并说明规则约束如何降低 AI 决策风险。",
  "supportingEvidence": [
    {
      "abstract": "Heads-up no-limit Texas hold’em (HUNL) is the quintessential game with imperfect information. Representative priorworks like DeepStack and Libratus heavily rely on counter-factual regret minimization (CFR) and its variants to tackleHUNL. However, the prohibitive computation cost of CFRiteration makes it difficult for subsequent researchers to learnthe CFR model in HUNL and apply it in other practical applications. In this work, we present AlphaHoldem, a high-performance and lightweight HUNL AI obtained with an end-to-end self-play reinforcement learning framework. The proposed framework adopts a pseudo-siamese architecture to directly learn from the input state information to the output actions by competing the learned model with its different historical versions. The main technical contributions include anovel state representation of card and betting information, amultitask self-play training loss function, and a new modelevaluation and selection metric to generate the final model.In a study involving 100,000 hands of poker, AlphaHoldemdefeats Slumbot and DeepStack using only one PC with threedays training. At the same time, AlphaHoldem only takes 2.9milliseconds for each decision-making using only a singleGPU, more than 1,000 times faster than DeepStack. We release the history data among among AlphaHoldem, Slumbot,and top human professionals in the author’s GitHub repository to facilitate further studies in this direction.",
      "evidence_level": "abstract_only",
      "first_published": "2022-06-28",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "AlphaHoldem在10万手牌研究中击败Slumbot和DeepStack，仅需一台PC训练三天。",
        "每次决策仅需2.9毫秒，比DeepStack快1000倍以上。"
      ],
      "method": "采用端到端自对弈强化学习框架，使用伪孪生架构直接从输入状态学习输出动作，引入了新的牌面和下注信息状态表示、多任务自对弈训练损失函数和模型评估选择指标。",
      "metrics": {
        "citation_count": 70,
        "discovered_via": "reference",
        "discovery_seed": "arxiv:2401.06168",
        "graph_bridge": 0.038249356381022434,
        "graph_citation_in_degree": 2,
        "graph_community": 0,
        "graph_foundation": 0.06870537698656023,
        "graph_frontier": 0.4968542149273121,
        "graph_reference_out_degree": 2,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "influential_citation_count": 7,
        "landscape_direction_similarity": 0.5182069383995007,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 35,
        "retrieval_lexical_score": 0.14705882352941174,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.49176243890211546,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "AAAI Conference on Artificial Intelligence"
      },
      "paper_id": "s2:4fcf18bda55414c5f31cc4be560bae92e8e4b7e9",
      "rank": 11,
      "relevance": "展示了在不完美信息博弈中通过自对弈强化学习实现高效决策的可行性，对理解人类在不确定性下的快速决策、风险管理和策略学习具有启示意义。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/4fcf18bda55414c5f31cc4be560bae92e8e4b7e9",
      "summary_status": "generated",
      "title": "AlphaHoldem: High-Performance Artificial Intelligence for Heads-Up No-Limit Poker via End-to-End Reinforcement Learning"
    },
    {
      "abstract": "Artificial intelligence (AI) has surpassed top human players in a variety of games. In imperfect information games, these achievements have primarily been driven by Counterfactual Regret Minimization (CFR) and its variants for computing Nash equilibrium. However, most existing research has focused on maximizing payoff, while largely neglecting the importance of strategic diversity and the need for varied play styles, thereby limiting AI's adaptability to different user preferences. To address this gap, we propose Preference-CFR (Pref-CFR), a novel method that incorporates two key parameters: preference degree and vulnerability degree. These parameters enable the AI to adjust its strategic distribution within an acceptable performance loss threshold, thereby enhancing its adaptability to a wider range of strategic demands. In our experiments with Texas Hold'em, Pref-CFR successfully trained Aggressive and Loose Passive styles that not only match original CFR-based strategies in performance but also display clearly distinct behavioral patterns. Notably, for certain hand scenarios, Pref-CFR produces strategies that diverge significantly from both conventional expert heuristics and original CFR outputs, potentially offering novel insights for professional players.",
      "evidence_level": "abstract_only",
      "first_published": "2024-11-02",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "在德州扑克实验中，Preference-CFR成功训练出激进和松散被动风格，性能与原始CFR策略相当但行为模式明显不同。",
        "在某些手牌场景中，产生的策略与传统专家启发式和原始CFR显著不同，可能为职业选手提供新见解。"
      ],
      "method": "提出Preference-CFR方法，在反事实遗憾最小化中引入偏好度和脆弱度两个参数，使智能体在可接受性能损失范围内调整策略分布。",
      "metrics": {
        "citation_count": 0,
        "citation_normalized_percentile": null,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.44312172271968214,
        "graph_reference_out_degree": 1,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "is_retracted": false,
        "landscape_direction_similarity": 0.5204953647717343,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv (Cornell University)"
      },
      "paper_id": "openalex:W4404351335",
      "rank": 15,
      "relevance": "说明在决策中考虑策略多样性和风险偏好的重要性，启示人类在风险管理中可以根据自身偏好调整决策风格，并在保持性能的同时探索非常规策略。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "http://arxiv.org/abs/2411.01217",
      "summary_status": "generated",
      "title": "Preference-CFR$\\:$ Beyond Nash Equilibrium for Better Game Strategies"
    },
    {
      "abstract": "Deciding which of two agents is stronger means playing games until skill outweighs luck, and every game costs money, model inference, or expert time. Since the number of games needed is unknown, fixed-budget evaluations either keep paying after the result is settled or stop before the agents can be told apart, while naive optional stopping with an ordinary confidence interval invalidates the stated level. We make such an evaluation stop as soon as its evidence suffices, with the guarantee intact. The Action-Informed Value Assessment Tool (AIVAT) reduces variance in imperfect-information games through conditional mean-zero corrections, by a median $54\\times$ across 15 LLM agent configurations spanning 71,439 paired Heads-Up No-Limit Hold'em (HUNL) hands, but does not say when to stop. We combine AIVAT with continuously monitored Confidence Sequences (CSs) into anytime-valid AIVAT (AV-AIVAT), whose online value model learns only from past games so that no game scores its own correction. At the nominal 95\\% level and a target precision of $\\pm1$ Big Blind, raw outcomes need a median $74\\times$ as many hands as AIVAT-corrected outcomes to stop under the Asymptotic CS (AsympCS). Exact finite-sample certification uses the Empirical-Bernstein CS (EB-CS), which needs an independently justified bound on corrected payoffs. We establish such a bound structurally for Leduc hold'em and characterize a width floor set by the CS's bet cap and that bound, which governs how much of a variance gain becomes earlier stopping; the descriptive HUNL EB-CS runs show a median $1.37\\times$ stopping-time ratio. AV-AIVAT turns variance reduction into efficient, auditable early stopping while separating asymptotic screening from exact certification, so an evaluation can stop the moment its evidence suffices and hand a third party everything needed to recheck the verdict at that very stopping time.",
      "evidence_level": "abstract_only",
      "first_published": "2026-08-06",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "在名义95%水平和±1大盲目标精度下，原始结果所需手数中位数是AIVAT校正结果的74倍。",
        "在Leduc hold'em中建立了精确有限样本认证，在HUNL描述性运行中实现中位数1.37倍的停止时间比。"
      ],
      "method": "将AIVAT方差缩减技术与随时有效置信序列结合，形成AV-AIVAT，通过在线值模型从过去游戏中学习并避免自我校正，使用渐近置信序列和实证Bernstein置信序列进行认证。",
      "metrics": {
        "citation_count": 0,
        "citation_counts": [
          0.0,
          1.0
        ],
        "citation_normalized_percentile": null,
        "discovered_via": "recommendation",
        "graph_bridge": 0.04656123574843693,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.8,
        "graph_reference_out_degree": 5,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "influential_citation_count": 0,
        "is_retracted": false,
        "landscape_direction_similarity": 0.3938339773388779,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 53,
        "retrieval_lexical_score": 0.029411764705882353,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.38663209680516475,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv (Cornell University)"
      },
      "paper_id": "openalex:W7202006369",
      "rank": 16,
      "relevance": "解决了在不完美信息博弈中评估智能体强弱时何时停止收集证据的问题，对现实决策中如何在不确定环境中高效地判断风险并做出及时结论具有启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://arxiv.org/abs/2608.06362",
      "summary_status": "generated",
      "title": "AV-AIVAT: 74x Cheaper Agent Evaluation with Certified Anytime-Valid Stopping in Imperfect-Information Games"
    },
    {
      "abstract": "We introduce and analyze Limit Continuous Poker, a variant of Von Neumann's Continuous Poker with variable but limited bet sizes. This simplified variant of poker captures aspects of information asymmetry, bluffing, balancing, and the impact of bet size limits while still being simple enough to solve analytically. We derive the Nash equilibrium strategy profile for this game, showing how the bettor's and caller's strategies depend on the bet size limits. We demonstrate that as the bet size limits approach extreme values, the strategy profile converges to those of other continuous poker variants. Finally, we connect these results to strategic implications of limited bet sizing in real-world poker.",
      "evidence_level": "abstract_only",
      "first_published": "2026-05-31",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "推导出限注连续扑克的纳什均衡策略，并揭示其如何随下注限制变化。",
        "当下注限制趋于极端值时，策略收敛到其他连续扑克变体，并将结论联系到现实扑克中有限下注的战略意义。"
      ],
      "method": "引入并分析限注连续扑克，推导该变体的纳什均衡策略，研究下注者与跟注者策略对下注大小限制的依赖关系。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.5283379417877712,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker"
      },
      "paper_id": "arxiv:2606.01390",
      "rank": 17,
      "relevance": "从理论层面剖析信息不对称、诈唬、平衡和注额限制的交互作用，为现实博弈中如何根据可用选项调整策略和风险管理提供数学基础。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2606.01390v1",
      "summary_status": "generated",
      "title": "Limit Continuous Poker: A Variant of Continuous Poker with Limited Bet Sizes"
    },
    {
      "abstract": "An agent playing a Nash-equilibrium strategy in a two-player zero-sum imperfect-information game secures the game value but forfeits the additional value offered by a flawed opponent. Diffuse deviations pose a particular challenge: binary release rules may gather too little evidence to act, while a full best response to an incomplete opponent model can be highly exploitable. We introduce \\emph{budget-constrained confidence-scheduled restricted responses} (CS-RNR), the first opponent-exploitation method whose safety guarantee is a certificate the agent computes on the strategy it actually deploys, so that every exploit it commits to is one it has audited itself. The method tracks pooled action frequencies with anytime-valid confidence sequences and treats a frequency as exploitable only once its interval separates from an equilibrium reference. The confirmed deviations define a conservative opponent model, which a restricted-response solve turns into candidate counter-strategies over a grid of pin levels. Before deployment, each complete candidate is evaluated by a full-tree best response. The resulting certificate is compared with a user-specified budget and committed atomically with the strategy. Because this check is performed on the played strategy, model quality determines the exploitation achieved while the certificate controls reference-relative expected loss. In Leduc hold'em, CS-RNR obtains $6.2\\times$ the steady-state gain of a money-verified binary gate while keeping every deployed strategy within budget. A trajectory mixture using the same estimator reaches $13.6\\times$ the budget. Across Leduc, Liar's Dice, and 5-rank Leduc, all $36{,}000$ audited hands satisfy the reported certificate tolerance.",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-30",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "在Leduc hold'em中，CS-RNR获得6.2倍于金钱验证二元门控的稳态收益，同时保持每个部署策略在预算内。",
        "在所有36000手审计手中均满足报告的证书容差，证明了方法的安全性和可靠性。"
      ],
      "method": "提出预算约束的置信度调度受限响应方法，使用随时有效置信序列跟踪对手动作频率，仅当置信区间与均衡参考分离时判定可剥削，通过受限响应求解生成候选策略，并用全树最佳响应在部署前评估生成安全证书。",
      "metrics": {
        "citation_count": 2,
        "discovered_via": "recommendation",
        "graph_bridge": 0.036925340198602434,
        "graph_citation_in_degree": 1,
        "graph_community": 0,
        "graph_foundation": 0.015274084327856138,
        "graph_frontier": 0.8091653098719107,
        "graph_reference_out_degree": 5,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "influential_citation_count": 1,
        "landscape_direction_similarity": 0.4076178635479744,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 53,
        "retrieval_lexical_score": 0.023529411764705882,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4346038738318714,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": ""
      },
      "paper_id": "s2:b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
      "rank": 18,
      "relevance": "展示了在不完美信息博弈中既利用对手弱点又严格控制风险的机制，对现实决策中如何在利用他人认知偏差的同时管理潜在损失具有重要启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/b23b9276a15a4ee0a3bf9af90548dc37c4513bcf",
      "summary_status": "generated",
      "title": "Agents That Certify Their Own Exploits: Confidence-Scheduled Restricted Responses for Safe Opponent Exploitation"
    },
    {
      "abstract": "We present StratFormer, a transformer-based meta-agent that learns to simultaneously model and exploit opponents in imperfect-information games through a two-phase curriculum. The first phase trains an opponent modeling head to identify behavioral patterns from action histories while the agent plays a game-theoretic optimal (GTO) policy. The second phase progressively shifts the policy toward best-response (BR) exploitation, guided by a per-opponent regularization schedule tied to exploitability. Our architecture introduces dual-turn tokens -- feature vectors constructed at both agent and opponent decision points -- coupled with bucket-rate features that encode opponent tendencies across five strategic contexts. On Leduc Hold'em, a small poker variant with six cards and two betting rounds, we test against six opponent archetypes at two strength levels each, with exploitability ranging from 0.15 to 1.26 Big Blinds (BB) per hand. StratFormer achieves an average exploitation gain of +0.106 BB per hand over GTO, with peak gains of +0.821 against highly exploitable opponents, while maintaining near-equilibrium safety.",
      "evidence_level": "abstract_only",
      "first_published": "2026-04-28",
      "limitations": [
        "仅在 Leduc Hold'em 小型扑克变体上验证，摘要未说明对大型扑克的泛化情况。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "StratFormer 能同时建模和利用多种对手行为，平均对 GTO 获得 +0.106 大盲注/手的利用增益，对高度可利用对手峰值增益 +0.821。",
        "该模型在提升利用效果的同时保持接近均衡安全性。"
      ],
      "method": "提出基于 Transformer 的元代理 StratFormer，采用两阶段课程学习：第一阶段在博弈论最优策略下训练对手建模头，第二阶段按可利用性驱动的正则化逐步转向最佳响应利用；架构使用双回合标记和桶率特征。",
      "metrics": {
        "citation_count": 0,
        "citation_normalized_percentile": null,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 1,
        "graph_community": 0,
        "graph_foundation": 0.017870678662483413,
        "graph_frontier": 0.5531217227196821,
        "graph_reference_out_degree": 1,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "is_retracted": false,
        "landscape_direction_similarity": 0.4553780012225313,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv (Cornell University)"
      },
      "paper_id": "openalex:W7159547552",
      "rank": 19,
      "relevance": "通过德州扑克变体展示自适应对手建模与利用的可行性，启示现实决策中应结合对他人行为的预测与自身策略的稳健性，在风险与收益间取得平衡。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://arxiv.org/abs/2604.25796",
      "summary_status": "generated",
      "title": "StratFormer: Adaptive Opponent Modeling and Exploitation in Imperfect-Information Games"
    },
    {
      "abstract": "We explore the creation of human models of poker players with a combination of deep neural networks and regularized search. We begin with behavioral cloning and explain our neural architecture choices. Next, we qualitatively show that our model can express a variety of human-like behaviors. Finally, we use the aforementioned model combined with regularized search to understand the tradeoff between human-likeness and policy strength in poker. In particular, we find that we can increase the human-likeness of a near-equilibrium policy for a low cost in exploitability, or increase the strength of a human-like policy for a low cost in human prediction accuracy. This thesis lays groundwork for using computers to create human-like opponents to train against.",
      "evidence_level": "abstract_only",
      "first_published": "2026-02-01",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "模型能够表达多种类似人类的行为。",
        "人类相似性与策略强度之间存在权衡，可以在低可利用性成本下提升近似均衡策略的人类相似性，或在低人类预测准确性成本下提升人类类似策略的强度。"
      ],
      "method": "结合深度神经网络与正则化搜索，以行为克隆为基础构建扑克玩家的人类模型。",
      "metrics": {
        "citation_count": 0,
        "citation_normalized_percentile": null,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "is_retracted": false,
        "landscape_direction_similarity": 0.546928306074921,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "DSpace@MIT (Massachusetts Institute of Technology)"
      },
      "paper_id": "openalex:W7202060719",
      "rank": 20,
      "relevance": "展示了如何用计算模型刻画人类打牌行为，有助于理解人类决策模式，并为在现实场景中训练更拟人化的智能体或评估人类行为偏差提供思路。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://dspace.mit.edu/handle/1721.1/171485",
      "summary_status": "generated",
      "title": "On Creating Human Models in Poker with Deep Learning and Regularized Search"
    },
    {
      "abstract": "orientador do(a",
      "evidence_level": "abstract_only",
      "first_published": "2026-08-08",
      "limitations": [
        "摘要未说明",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "摘要未说明"
      ],
      "method": "摘要未说明",
      "metrics": {
        "citation_count": 0,
        "citation_normalized_percentile": null,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "is_retracted": false,
        "landscape_direction_similarity": 0.6057261505667366,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": null
      },
      "paper_id": "openalex:W7203644380",
      "rank": 21,
      "relevance": "标题指向使用大语言模型学习战略扑克决策，可能与利用语言模型改进决策相关，但摘要未提供实质内容。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://doi.org/10.14393/ufu.di.2026.510",
      "summary_status": "generated",
      "title": "Learning strategic poker decision-making with Large Language Models"
    },
    {
      "abstract": "Humans exhibit irrational decision-making patterns in response to environmental triggers, such as experiencing an economic loss or gain. In this paper we investigate whether algorithms exhibit the same behavior by examining the observed decisions and latent risk and rationality parameters estimated by a random utility model with constant relative risk-aversion utility function. We use a dataset consisting of 10,000 hands of poker played by Pluribus, the first algorithm in the world to beat professional human players and find (1) Pluribus does shift its playing style in response to economic losses and gains, ceteris paribus; (2) Pluribus becomes more risk-averse and rational following a trigger but the humans become more risk-seeking and irrational; (3) the difference in playing styles between Pluribus and the humans on the dimensions of risk-aversion and rationality are particularly differentiable when both have experienced a trigger. This provides support that decision-making patterns could be used as \"behavioral signatures\" to identify human versus algorithmic decision-makers in unlabeled contexts.",
      "evidence_level": "abstract_only",
      "first_published": "2021-11-14",
      "limitations": [
        "分析基于 Pluribus 这一单一 AI 系统及其 10,000 手扑克数据，摘要未说明对其他算法的普遍性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "Pluribus 在控制其他因素后会对经济得失调整打牌风格。",
        "经历触发后，Pluribus 变得更风险厌恶和理性，而人类变得更风险寻求和非理性。",
        "在双方都经历触发后，Pluribus 与人类在风险厌恶和理性维度上的风格差异尤其明显，可作为“行为特征”区分算法与人类决策者。"
      ],
      "method": "使用常数相对风险厌恶效用函数的随机效用模型，估计 Pluribus 与人类玩家的潜在风险和理性参数，并比较触发经济损失/收益前后的决策。",
      "metrics": {
        "graph_bridge": 0.1588819418904009,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.321825840795231,
        "graph_reference_out_degree": 2,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "landscape_direction_similarity": 0.5143182486852238,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games"
      },
      "paper_id": "arxiv:2111.07295",
      "rank": 22,
      "relevance": "直接对比人工智能与人类在扑克中的损失/收益反应，揭示人类在风险决策中的非理性倾向，有助于理解认知偏差并设计更稳健的决策系统。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2111.07295v1",
      "summary_status": "generated",
      "title": "Rational AI: A comparison of human and AI responses to triggers of economic irrationality in poker"
    },
    {
      "abstract": "Strategic reasoning under uncertainty underpins consequential decisions in negotiation, finance, and policy, but prevailing game-play benchmarks collapse heterogeneous reasoning dimensions into a single scalar, leaving the capability structure of frontier LLMs unexamined. We introduce Poker Arena, a no-limit Texas Hold'em tournament platform that couples a three-layer memory architecture (within-hand, session, and cross-session) with a nine-axis cognitive profile decomposing strategic reasoning into interpretable dimensions such as bet-sizing calibration and positional awareness. We evaluate seven frontier models across 50 sessions of 1,000 hands and a controlled memory ablation; tournament chips and aggregate axis score order the field differently: Claude Opus 4.6 wins +$15,730 chips with 14 first-place finishes, yet ranks only fifth of seven on mean axis score, while persistent memory helps some models and hurts others. These findings show that multi-axis evaluation surfaces capability structure that scalar leaderboards systematically misrank, with cross-dimensional consistency outweighing peak performance on any single axis.",
      "evidence_level": "abstract_only",
      "first_published": "2026-06-11",
      "limitations": [
        "评估对象仅限前沿 LLM，未包含人类玩家；摘要未说明其他局限。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "锦标赛筹码排名与聚合轴得分排名不一致，如 Claude Opus 4.6 赢得最多筹码但平均轴得分仅排第五。",
        "多轴评估能揭示单一标量排行榜系统错排的能力结构，跨维度一致性比单轴峰值更重要。",
        "持久记忆对部分模型表现有帮助，对另一些反而有害。"
      ],
      "method": "构建 Poker Arena 无限注德州扑克平台，采用三层记忆架构（手内、会话、跨会话）和九轴认知画像，对七个前沿 LLM 进行 50 个会话（每会话 1000 手）的评估及受控记忆消融。",
      "metrics": {
        "citation_count": 0,
        "citation_normalized_percentile": null,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.558272565207101,
        "graph_reference_out_degree": 2,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "is_retracted": false,
        "landscape_direction_similarity": 0.5263633058655623,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv (Cornell University)"
      },
      "paper_id": "openalex:W7164973936",
      "rank": 23,
      "relevance": "通过德州扑克评估大语言模型的战略推理和记忆能力，说明在现实决策中应采用多维评估而非单一指标，并关注记忆与认知结构的相互作用。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://arxiv.org/abs/2606.13815",
      "summary_status": "generated",
      "title": "Poker Arena: Multi-Axis Profiling of Strategic Reasoning and Memory in LLMs"
    },
    {
      "abstract": "Poker, a million-dollar industry and a leisure activity, is played between close-knit friends, at casinos, and online. It is gendered; masculine positions are produced between players, often in competitive, aggressive, and sexist ways. However, the (homo)social relationships of poker have hardly been studied. Using interviews with Swedish men poker players, this article investigates masculine positions and homosocialities within poker, demonstrating that poker was associated with a range of social relationships, from close friendships to relatively anonymous online relationships. Poker made room for friendship, but it also structured relationships and led to a simplification and a formalization of them. Competition and emotional detachment were central, but not sexism; the article suggests that poker players are constructed as nerdy, rather than tough guys, and that nerdy socialities are formed between players.",
      "evidence_level": "abstract_only",
      "first_published": "2024-08-28",
      "limitations": [
        "样本仅限瑞典男性玩家，可能不具广泛代表性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "扑克关联多种社会关系，从亲密友谊到相对匿名的在线关系。",
        "扑克会结构化人际关系，使其简化和形式化。",
        "竞争与情感疏离是核心特征，玩家被构建为“书呆子”而非“硬汉”，并形成书呆子式社交。"
      ],
      "method": "使用访谈法，对瑞典男性扑克玩家进行质性研究，分析扑克中的男性位置与同性社交关系。",
      "metrics": {
        "citation_count": 3,
        "citation_normalized_percentile": 0.81283479,
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.5065809586904276,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "is_retracted": false,
        "landscape_direction_similarity": 0.4683909321448307,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "The Journal of Men s Studies"
      },
      "paper_id": "openalex:W4402025026",
      "rank": 24,
      "relevance": "从社会学角度揭示扑克活动中的社会互动和情感规则，补充了对人类决策中社会因素和性别身份的理解，但未直接讨论认知偏差。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "openalex",
      "source_url": "https://doi.org/10.1177/10608265241279370",
      "summary_status": "generated",
      "title": "Play the Man, Not the Cards. (Homo)socialities and Masculine Positions in Poker"
    },
    {
      "abstract": "There has been significant recent progress in algorithms for approximation of Nash equilibrium in large two-player zero-sum imperfect-information games and exact computation of Nash equilibrium in multiplayer strategic-form games. While counterfactual regret minimization and fictitious play are scalable to large games and have convergence guarantees in two-player zero-sum games, they do not guarantee convergence to Nash equilibrium in multiplayer games. Recently, an approach has been presented for exact computation of Nash equilibrium in multiplayer imperfect-information games that solves a quadratically constrained program based on a nonlinear complementarity problem formulation derived from the sequence-form game representation. This formulation was solved using Gurobi's nonconvex quadratic solver, which employs spatial branch-and-bound to iteratively refine variable bounds by solving convex relaxations of bilinear terms via McCormick envelopes. During presolve, Gurobi introduces auxiliary variables and, in some cases, binary variables, leading to an internal MIQCP reformulation. This approach was demonstrated to outperform prior algorithms from the Gambit software suite and quickly solve three-player Kuhn poker after removal of dominated actions; however, the algorithm was not able to solve the full version of the game within 24 hours. In this paper, we derive finite bounds on slack and multiplier variables in the nonlinear complementarity formulation. These bounds strengthen the convex relaxations used within spatial branch-and-bound and lead to substantial computational improvements. We demonstrate the impact of the proposed bounds on exact Nash equilibrium computation in three-player Kuhn poker.",
      "evidence_level": "abstract_only",
      "first_published": "2026-06-24",
      "limitations": [
        "实验仅涉及小型三人 Kuhn 扑克，摘要未说明对完整版本游戏的适用性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "提出的界能显著提升精确纳什均衡计算的求解效率，成功用于三人 Kuhn 扑克。"
      ],
      "method": "在多人不完全信息博弈的精确纳什均衡计算中，推导非线性互补公式中松弛变量和乘子变量的有限界，强化空间分支定界的凸松弛。",
      "metrics": {
        "citation_count": 1,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 1,
        "graph_community": 0,
        "graph_foundation": 0.07637042163928066,
        "graph_frontier": 0.6076608016342326,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 4,
        "influential_citation_count": 1,
        "landscape_direction_similarity": 0.39594370755674396,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 10,
        "retrieval_lexical_score": 0.11764705882352941,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.41069062637738185,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv.org"
      },
      "paper_id": "s2:580ef7ab46739661ac4153a882adbb8e0753f5d2",
      "rank": 25,
      "relevance": "属于扑克博弈求解的算法优化，为理解多人不完全信息下的均衡计算提供工具，但对现实决策认知的直接启示有限。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/580ef7ab46739661ac4153a882adbb8e0753f5d2",
      "summary_status": "generated",
      "title": "Variable Bound Tightening for Nash Equilibrium Computation in Multiplayer Imperfect-Information Games"
    },
    {
      "abstract": "Deception has been a cornerstone of card games throughout human history, but what underlying patterns drive player behavior in social deduction games? This study explores player psychology and decision-making in the card game COUP using self-collected telemetry data from eight gameplay sessions. We identify key patterns shaped by risk sensitivity, uncertainty aversion, and emotional responses such as revenge. Despite being permitted and seemingly central to victory, in our small sample study, we discovered that lying was far less common than expected. Most players avoided bluffing whenever possible, and when they did bluff, they failed more often than they succeeded. Instead of pure strategic calculation, we found gameplay driven by emotional grudges, revenge spirals, and what we came to call the “hatred meter”: players targeting opponents who had wronged them, even when it defied optimal strategy. And we found meta-gaming patterns emerging across sessions, where players developed implicit tactical rules that never appeared in the official rulebook but spread through experiential learning. Our findings reveal how COUP serves as a compelling window into human decision-making, where risk aversion, cognitive load, perceived scarcity, and emotional dynamics could override rational strategic calculation in socially dynamic, competitive environments.",
      "evidence_level": "abstract_only",
      "first_published": "2026-06-29",
      "limitations": [
        "样本量小（8 个游戏会话），结果可能不具普遍性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "玩家在游戏中撒谎的频率低于预期，且撒谎时失败多于成功。",
        "情绪因素如仇恨、复仇螺旋常压倒理性策略，玩家会针对曾伤害自己的对手。",
        "会出现非官方规则中的元游戏模式，通过经验学习传播。"
      ],
      "method": "采用自收集的遥测数据，对 8 局 COUP 纸牌游戏进行观察分析，识别玩家心理与决策模式。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.5180190426378851,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 0,
        "retrieval_lexical_score": 0.07647058823529411,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.49895007236736144,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "Analog Game Studies"
      },
      "paper_id": "s2:4f3f90348d4117212e2c8a982327c1c61ea62cce",
      "rank": 26,
      "relevance": "通过牌类游戏验证风险敏感性、不确定性厌恶和情绪反应对人类决策的影响，说明现实决策中情绪和社会动态可能压倒纯理性计算。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/4f3f90348d4117212e2c8a982327c1c61ea62cce",
      "summary_status": "generated",
      "title": "Player Psychology and Decision-Making in Board Game COUP"
    },
    {
      "abstract": "We introduce PokerBench - a benchmark for evaluating the poker-playing abilities of large language models (LLMs). As LLMs excel in traditional NLP tasks, their application to complex, strategic games like poker poses a new challenge. Poker, an incomplete information game, demands a multitude of skills such as mathematics, reasoning, planning, strategy, and a deep understanding of game theory and human psychology. This makes Poker the ideal next frontier for large language models. PokerBench consists of a comprehensive compilation of 11,000 most important scenarios, split between pre-flop and post-flop play, developed in collaboration with trained poker players. We evaluate prominent models including GPT-4, ChatGPT 3.5, and various Llama and Gemma series models, finding that all state-of-the-art LLMs underperform in playing optimal poker. However, after fine-tuning, these models show marked improvements. We validate PokerBench by having models with different scores compete with each other, demonstrating that higher scores on PokerBench lead to higher win rates in actual poker games. Through gameplay between our fine-tuned model and GPT-4, we also identify limitations of simple supervised fine-tuning for learning optimal playing strategy, suggesting the need for more advanced methodologies for effectively training language models to excel in games. PokerBench thus presents a unique benchmark for a quick and reliable evaluation of the poker-playing ability of LLMs as well as a comprehensive benchmark to study the progress of LLMs in complex game-playing scenarios.",
      "evidence_level": "abstract_only",
      "first_published": "2025-01-14",
      "limitations": [
        "摘要提及监督微调的局限性，但未全面说明其他局限。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "所有最先进的大语言模型在最优扑克玩法上表现不足。",
        "微调后模型表现显著提升。",
        "PokerBench得分越高，在真实扑克游戏中的胜率越高；简单监督式微调难以学到最优策略。"
      ],
      "method": "构建含11,000个重要场景的PokerBench基准，场景分为翻牌前与翻牌后，与训练有素的扑克玩家合作设计；评估GPT-4、ChatGPT 3.5及多种Llama和Gemma系列模型，并通过对战验证分数与胜率的关系。",
      "metrics": {
        "graph_bridge": 0.0051489518205222505,
        "graph_citation_in_degree": 3,
        "graph_community": 0,
        "graph_foundation": 0.0605025572811503,
        "graph_frontier": 0.49500000000000005,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "landscape_direction_similarity": 0.4568352739042328,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker"
      },
      "paper_id": "arxiv:2501.08328",
      "rank": 27,
      "relevance": "该基准直接评估大语言模型在德州扑克等不完全信息游戏中的策略能力，其发现表明现有AI在类似人类决策的复杂博弈中仍有不足，也为利用AI辅助决策与策略训练提供了评估工具和启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2501.08328v2",
      "summary_status": "generated",
      "title": "PokerBench: Training Large Language Models to become Professional Poker Players"
    },
    {
      "abstract": "Conversational AI increasingly supports everyday decision-making, yet most systems rely on data-centric reasoning rather than the heuristic and interactional strategies people use in natural conversation. To ground design in actual human practice, we analyze 955 real-world Korean conversations (15,476 utterances) involving food and travel decisions, applying a decision-making codebook through an LLM-assisted coding pipeline. Our findings reveal that people prioritize satisficing over optimization, relying heavily on internal knowledge and interactional strategies to manage cognitive load. Critically, we identify a frequency-efficiency mismatch: the most prevalent heuristics sustain conversational flow during exploration, whereas infrequent, rule-based strategies are highly effective at driving resolution during exploitation. By mapping how these patterns transfer across the spectrum of human-AI interaction, this work provides empirical grounding consistent with cognitive theories of decision-making and offers design implications that align AI systems with human heuristic processes.",
      "evidence_level": "abstract_only",
      "first_published": "2026-05-08",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "人们更倾向于“满意即足够”而非最优化决策。",
        "最常见的启发式策略用于维持探索阶段的对话流畅，而罕见的规则型策略在利用阶段高效推动决策。",
        "存在启发式使用频率与决策效率之间的错配。"
      ],
      "method": "对955个真实韩语对话（15,476个话语）进行决策编码，借助LLM辅助的编码流水线分析食物与旅行决策中的启发式和策略。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 2,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.407351517101192,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 2,
        "topic_cluster_coherence": 0.6572789660170857,
        "topic_cluster_relation": "contextual",
        "topic_cluster_separation": 0.21417907310712875,
        "topic_cluster_title": "Cognitive Bias Amplification and Mitigation in AI-Assisted Choice"
      },
      "paper_id": "arxiv:2605.07789",
      "rank": 28,
      "relevance": "该研究揭示了日常决策中人类启发式与认知负荷管理的实际模式，为将德州扑克等博弈中的策略思维迁移到现实决策、并设计更符合人类启发式的AI辅助决策系统提供了实证基础。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2605.07789v1",
      "summary_status": "generated",
      "title": "Analyzing Human Heuristics and Strategies in Everyday Decision-Making Conversations for Conversational AI Design"
    },
    {
      "abstract": "We examine whether large language models (LLMs) can predict biased decision-making in conversational settings, and whether their predictions capture not only human cognitive biases but also how those effects change under cognitive load. In a pre-registered study (N = 1,648), participants completed six classic decision-making tasks via a chatbot with dialogues of varying complexity. Participants exhibited two well-documented cognitive biases: the Framing Effect and the Status Quo Bias. Increased dialogue complexity resulted in participants reporting higher mental demand. This increase in cognitive load selectively, but significantly, increased the effect of the biases, demonstrating the load-bias interaction. We then evaluated whether LLMs (GPT-4, GPT-5, and open-source models) could predict individual decisions given demographic information and prior dialogue. While results were mixed across choice problems, LLM predictions that incorporated dialogue context were significantly more accurate in several key scenarios. Importantly, their predictions reproduced the same bias patterns and load-bias interactions observed in humans. Across all models tested, the GPT-4 family consistently aligned with human behavior, outperforming GPT-5 and open-source models in both predictive accuracy and fidelity to human-like bias patterns. These findings advance our understanding of LLMs as tools for simulating human decision-making and inform the design of conversational agents that adapt to user biases.",
      "evidence_level": "abstract_only",
      "first_published": "2026-01-16",
      "limitations": [
        "摘要提及不同决策任务的预测结果不一致，但未明确其他局限性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "参与者表现出框架效应和现状偏见这两种经典认知偏见。",
        "对话复杂度增加导致报告的心智负荷上升，且认知负荷显著放大了偏见影响。",
        "GPT-4系列在预测准确性和人类偏见模式保真度上均优于其他测试模型。"
      ],
      "method": "预注册实验（N=1,648），让参与者通过聊天机器人完成六个经典决策任务，并变化对话复杂度以操纵认知负荷；随后评估多种LLM（GPT-4、GPT-5及开源模型）对个体决策的预测能力。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 2,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.40192919825655204,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 2,
        "topic_cluster_coherence": 0.6572789660170857,
        "topic_cluster_relation": "contextual",
        "topic_cluster_separation": 0.21417907310712875,
        "topic_cluster_title": "Cognitive Bias Amplification and Mitigation in AI-Assisted Choice"
      },
      "paper_id": "arxiv:2601.11049",
      "rank": 29,
      "relevance": "该研究直接证明认知负荷会放大决策偏见，并验证LLM能模拟人类偏见模式，这对理解德州扑克等高风险决策中的情绪与认知因素、以及设计适应性AI辅助决策工具有重要参考价值。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2601.11049v2",
      "summary_status": "generated",
      "title": "Predicting Biased Human Decision-Making with Large Language Models in Conversational Settings"
    },
    {
      "abstract": "Abstract Since the birth of the bounded rationality concept, scholars have been increasingly involved in identifying how management decisions are made; identifying the role played by affect has been a crucial part of the mission. However, despite the recent hype in the number of researches published on this theme – that has been carried into the new definition of bounded emotionality – a systematisation of those contributions able to identify the different functions played by different affective states is still lacking. This review article aims to fill this gap. The implemented methodology is the Systematic Literature Review. A total of 123 articles have been analysed through a descriptive as well as a thematic approach; the latter has followed a mixed inductive-deductive method. Results of the thematic analysis show six distinct functions played by affect in management decisions, offering an updated framework. The proposed model explains how affect influences management decisions on the basis of co-evolutionary mechanisms. The value of this work lies in offering a model of the functions played by affect in management decisions, which is pivotal for designing a firm's informed decision architecture. Furthermore, it is the first review on this theme to adopt a scientific approach and an established psychological framework for, respectively, the systematisation and analysis of the contributions.",
      "evidence_level": "abstract_only",
      "first_published": "2019-02-01",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "识别出情感在管理决策中发挥的六种不同功能。",
        "提出一个基于共同进化机制的解释模型，说明情感如何影响管理决策。"
      ],
      "method": "采用系统文献综述法，对123篇文章进行描述性分析和主题分析，主题分析采用混合归纳-演绎路径。",
      "metrics": {
        "citation_count": 80,
        "discovered_via": "reference",
        "discovery_seed": "arxiv:2111.07295",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 1,
        "graph_community": 1,
        "graph_foundation": 0.03818521081964033,
        "graph_frontier": 0.29424527211918994,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "influential_citation_count": 3,
        "landscape_direction_similarity": 0.385350151015326,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 102,
        "retrieval_lexical_score": 0.12352941176470587,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.3817379619779436,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": "European Management Journal"
      },
      "paper_id": "s2:2f1085d977583f282d29e11ef8aef33307e36d07",
      "rank": 30,
      "relevance": "该综述系统化了情感在管理决策中的作用，为将德州扑克中的情绪管理与风险决策研究拓展到组织管理场景提供了理论框架，有助于设计“知情决策架构”。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/2f1085d977583f282d29e11ef8aef33307e36d07",
      "summary_status": "generated",
      "title": "The role of affect in management decisions: A systematic review"
    },
    {
      "abstract": "Poker is a multiplayer game of imperfect information and has been widely studied in game theory. Many popular variants of poker (e.g., Texas Hold'em and Omaha) at the edge of modern game theory research are large games. However, even toy poker games, such as Kuhn poker, can pose new challenges. Many Kuhn poker variants have been investigated: varying the number of players, initial pot size, and number of betting rounds. In this paper we analyze a new variant -- Kuhn poker with cheating and cheating detection. We determine how cheating changes the players' strategies and derive new analytical results.",
      "evidence_level": "abstract_only",
      "first_published": "2020-11-09",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "作弊会改变玩家的博弈策略。",
        "作者推导出新的解析结果，刻画了作弊存在时的策略变化。"
      ],
      "method": "对Kuhn扑克的新变型（引入作弊与作弊检测）进行理论分析，推导关于策略变化的新解析结果。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.22000000000000003,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.5592589637012488,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker"
      },
      "paper_id": "arxiv:2011.04450",
      "rank": 31,
      "relevance": "该研究为不完全信息博弈中欺骗与检测的建模提供了理论基础，对理解德州扑克等游戏中的欺骗风险和应对策略具有直接意义，并可延伸至现实中的信息不对称决策。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2011.04450v1",
      "summary_status": "generated",
      "title": "Kuhn Poker with Cheating and Its Detection"
    },
    {
      "abstract": "Many important games have more than two players and imperfect information. Existing approaches for computing Nash equilibrium, the central game-theoretic solution concept, in such games either lack scalability or obtain poor performance. In this paper we introduce a new algorithm called projected exploitability descent (PED) for approximating Nash equilibria in multiplayer games of imperfect information. The algorithm works by running projected subgradient descent minimizing a proxy for the multiplayer generalized exploitability function. The objective is nonconvex and nonsmooth, but can be represented as the sum of the maxima of linear functions, for which a subgradient can easily be computed and projected to the polytope of feasible sequence-form strategies. We explore performance of PED on a generalized version of the well-studied benchmark game three-player Kuhn poker. No prior exact algorithms scale to the version of the game with deck size larger than 4, and we compare performance to the popular algorithms of fictitious play (FP) and counterfactual regret minimization (CFR). We find that PED obtains a consistent near-monotonic improvement throughout all runs, though both FP and CFR perform significantly better in the initial iterations. This inspires a hybrid algorithm FP-PED that runs FP for an initial burn-in period before switching to PED for stable long-run refinement. We can alternatively view this as a multi-step algorithm that runs FP as a pre-processing step to obtain a strong initialization for PED.",
      "evidence_level": "abstract_only",
      "first_published": "2026-06-28",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "PED在运行过程中表现出持续的近单调改进。",
        "在初始迭代中FP和CFR显著优于PED，但FP-PED混合算法结合了FP的快速初始化和PED的长期稳定改进。"
      ],
      "method": "提出投影可利用性下降（PED）算法，通过投影次梯度下降最小化多人广义可利用性的代理函数；在三玩家Kuhn扑克基准上实验，并与虚拟博弈（FP）和反事实遗憾最小化（CFR）比较。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.593704118075549,
        "graph_reference_out_degree": 1,
        "graph_structural_role": "frontier",
        "graph_topology_community": 4,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.4269436803238345,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 17,
        "retrieval_lexical_score": 0.12352941176470587,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4244021038940245,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": "arXiv.org"
      },
      "paper_id": "s2:fb7b874fecd6f6ac2218e949d3431b097d88a2b5",
      "rank": 32,
      "relevance": "该算法面向多人不完美信息博弈（如德州扑克多人局）的纳什均衡计算，为扑克AI策略求解提供了新的数值方法，也启示了现实中多主体竞争情境下的均衡思维。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/fb7b874fecd6f6ac2218e949d3431b097d88a2b5",
      "summary_status": "generated",
      "title": "Projected Exploitability Descent for Nash Equilibrium Computation in Multiplayer Imperfect-Information Games"
    },
    {
      "abstract": "Mixed strategy equilibrium predicts i.i.d play: past actions should not help predict future decisions. Human players, however, systematically depart from this benchmark, and in O'Neill's zero sum card game, these departures can be predicted by black box sequence models such as LSTMs. This paper asks whether that predictive power can be achieved by transparent alternatives that also reveal the behavioural structure behind it. Using 84,060 decisions from 2,802 pairs, the analysis first benchmarks naive and behavioral models against interpretable machine learning and deep learning models, then evaluates the modified EWA specifications of prior work against these benchmarks and uses the LASSO diagnostics to motivate a further nested frequency tracking extension. The results show that repeat or avoid behavior, especially players'management of their own recent action histories, accounts for most of the interpretable and strategically exploitable signal, while frequency tracking adds little out of sample.",
      "evidence_level": "abstract_only",
      "first_published": "2026-08-07",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "重复或避免行为（尤其是玩家对自己近期动作历史的管理）占据了大部分可解释且战略可利用的信号。",
        "频率跟踪在样本外预测中的增量贡献很小。"
      ],
      "method": "利用84,060个决策（来自2,802对玩家）的数据集，对朴素模型、行为模型、可解释机器学习模型和深度学习模型进行基准比较；评估改进的EWA模型，并用LASSO诊断提出嵌套频率跟踪扩展。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.4709135493255879,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 12,
        "retrieval_lexical_score": 0.041176470588235294,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4610731574269699,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": ""
      },
      "paper_id": "s2:c2388f1811e05ef63ae6ce0a862faa3365fda3b8",
      "rank": 33,
      "relevance": "该研究揭示了人类在博弈中偏离混合策略均衡的系统性模式，为从行为经济学角度理解德州扑克等游戏中的决策偏差、并发展可解释的博弈策略模型提供了证据。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/c2388f1811e05ef63ae6ce0a862faa3365fda3b8",
      "summary_status": "generated",
      "title": "Beyond the Black Box: Interpretable Models of Human Randomisation Failures"
    },
    {
      "abstract": "Cognitive biases are systematic errors in judgment. Researchers in data visualizations have explored whether cognitive biases transfer to decision-making tasks with interactive data visualizations. At the same time, cognitive scientists have reinterpreted cognitive biases as the product of resource-rational strategies under finite time and computational costs. In this paper, we argue for the integration of resource-rational analysis through constrained Bayesian cognitive modeling to understand cognitive biases in data visualizations. The benefit would be a more realistic \"bounded rationality\" representation of data visualization users and provides a research roadmap for studying cognitive biases in data visualizations through a feedback loop between future experiments and theory",
      "evidence_level": "abstract_only",
      "first_published": "2020-09-28",
      "limitations": [
        "摘要未说明。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "提出用资源理性分析重新解释认知偏见，为可视化用户建立更现实的“有界理性”表征。",
        "给出了研究路线图，通过实验与理论的反馈循环研究数据可视化中的认知偏见。"
      ],
      "method": "采用概念性论证，提出将资源理性分析（通过约束贝叶斯认知建模）整合到交互式数据可视化研究中，以解释认知偏见。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 2,
        "graph_foundation": 0.0,
        "graph_frontier": 0.22000000000000003,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.40593146201638436,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 2,
        "topic_cluster_coherence": 0.6572789660170857,
        "topic_cluster_relation": "contextual",
        "topic_cluster_separation": 0.21417907310712875,
        "topic_cluster_title": "Cognitive Bias Amplification and Mitigation in AI-Assisted Choice"
      },
      "paper_id": "arxiv:2009.13368",
      "rank": 34,
      "relevance": "该研究为理解人类在信息展示下的决策偏见提供了理论框架，与德州扑克策略中的有限理性、信息处理约束和风险判断直接相关，可用于改进决策辅助工具的设计。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2009.13368v2",
      "summary_status": "generated",
      "title": "Using Resource-Rational Analysis to Understand Cognitive Biases in Interactive Data Visualizations"
    },
    {
      "abstract": "Behavioral decision-making tasks are widely used to study individual and social preferences, including risk-taking, temporal discounting, and cooperation-related choices such as social value orientation, trust, and reciprocity, as well as more complex social behaviors examined through social dilemma and coordination games. These tasks are often administered along with typical survey questionnaires. However, scripting complexity limits their implementation on some popular online survey platforms (e.g., Qualtrics), which are commonly used to deploy studies across large populations. Here, we share a detailed experimental protocol and the corresponding source code, which enable the modular implementation of a decision-making battery in Qualtrics, including tasks such as the social value orientation, prisoner’s dilemma, trust game, and baseline nonsocial risk preference tasks. Data from 392 participants in Singapore and 94 participants in the United States (US) were collected and analyzed to validate the task battery. Their responses exhibited good quality and high convergent and divergent validity across different tasks and aligned with basic predictions of behavioral decision theory (e.g., the reflection effect and loss aversion) and social decision-making theories (e.g., inequity aversion and betrayal aversion). Responses from 314 Singapore participants and 94 US participants are shared (with their consent). The source code for the task batteries could be used to expand the database and for hypothesis testing in decision-making research. The data could facilitate comparisons with other populations and the development of simulated agents to address logistical challenges in asynchronous experiments. Overall, we go beyond mere code and data sharing to foster large-scale behavioral game theory research.",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-06",
      "limitations": [
        "摘要未说明样本代表性或文化差异的局限性。",
        "摘要未说明任务电池在复杂博弈情境中的效度。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "被试反应质量良好，不同任务间表现出较高的收敛效度和区分效度，并符合行为决策理论（如反射效应、损失厌恶）和社会决策理论（如不平等厌恶、背叛厌恶）的基本预测。"
      ],
      "method": "开发了Qualtrics平台上行为博弈论任务模块化实施的详细实验方案和源代码，包含社会价值取向、囚徒困境、信任博弈及基线非社会风险偏好任务，并用新加坡和美国被试数据验证任务电池的有效性。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.40036477447229407,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 115,
        "retrieval_lexical_score": 0.03529411764705882,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.39869299786867907,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": "Behavior Research Methods"
      },
      "paper_id": "s2:b7d662729e6a5dbc0cdc1f0fb4c08c04942abd11",
      "rank": 36,
      "relevance": "该研究验证了风险决策和行为博弈任务的有效测量工具，为研究人类决策与风险管理提供了可靠的实验范式，但对德州扑克策略本身未涉及，可视为决策心理学方法论的间接支撑。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/b7d662729e6a5dbc0cdc1f0fb4c08c04942abd11",
      "summary_status": "generated",
      "title": "QualGames: A Qualtrics implementation and a database of behavioral game theory tasks"
    },
    {
      "abstract": "This paper provides the foundations of a unified cognitive decision-making framework (QulBIT) which is derived from quantum theory. The main advantage of this framework is that it can cater for paradoxical and irrational human decision making. Although quantum approaches for cognition have demonstrated advantages over classical probabilistic approaches and bounded rationality models, they still lack explanatory power. To address this, we introduce a novel explanatory analysis of the decision-maker's belief space. This is achieved by exploiting quantum interference effects as a way of both quantifying and explaining the decision-maker's uncertainty. We detail the main modules of the unified framework, the explanatory analysis method, and illustrate their application in situations violating the Sure Thing Principle.",
      "evidence_level": "abstract_only",
      "first_published": "2020-05-30",
      "limitations": [
        "摘要未说明该框架在实证应用中的具体验证范围。",
        "摘要未提及该框架在实时决策或博弈对抗中的可操作性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "该框架能够处理人类决策中的悖论和非理性行为，比经典概率方法和有限理性模型更具解释力。"
      ],
      "method": "提出了基于量子理论的统一认知决策框架QuLBIT，通过量子干涉效应量化和解释决策者的不确定性，并应用于违反确定原则（Sure Thing Principle）的情境。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.22000000000000003,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.3606650350228301,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games"
      },
      "paper_id": "arxiv:2006.02256",
      "rank": 37,
      "relevance": "该研究为理解人类在风险决策中的非理性选择提供了新解释框架，可能有助于揭示德州扑克中玩家违反理性原则的认知机制。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2006.02256v2",
      "summary_status": "generated",
      "title": "QuLBIT: Quantum-Like Bayesian Inference Technologies for Cognition and Decision"
    },
    {
      "abstract": "In contrast to binary distinctions between outcomes such as wins and losses, more fine-grained distinctions include the representation of goal proximity, such as being near or far to desired or undesired outcomes. Despite semantic inconsistencies in near-win and near-loss literature, we offer a resolution where a near-loss is defined as an actual positive outcome described in reference to a negative counterfactual. Furthermore, we leverage the commercial \"push-your-luck\" game Can't Stop (Sackson, 1980) as a playful paradigm to study the behavioural consequences of near-misses compared to full-wins, playing against opponents who are either cautious or reckless. Each turn, players are given the option to either continue rolling dice (but risk losing their progress) or stop (thereby saving their progress). Our data showed that previous outcome and opponency interacted such that participants produced more stop behaviour against the cautious opponent relative to the reckless opponent but only when the previous outcome was a near-loss. Imitation of an opponent's behaviour is discussed both in reference to its potential automaticity and social contingencies, whereas stopping behaviour is considered a possibly erroneous perception of the interdependence between dice throws. (PsycInfo Database Record (c) 2026 APA, all rights reserved).",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-20",
      "limitations": [
        "摘要未说明样本量或统计效力。",
        "摘要未说明近失败定义的普适性在其他博弈中的适用性。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "先前结果与对手风格存在交互作用：仅当先前结果为近失败时，面对谨慎对手比面对鲁莽对手产生更多停止行为。",
        "对手行为的模仿可能与自动性和社会情境有关，停止行为可能源于对掷骰间相互依赖的错误感知。"
      ],
      "method": "使用商业“冒险”游戏Can't Stop作为实验范式，操纵先前结果（近失败 vs 完全胜利）和对手风格（谨慎 vs 鲁莽），测量玩家停止或继续掷骰的行为。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.45405339069586176,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 0,
        "retrieval_lexical_score": 0.023529411764705882,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4142124426464282,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": "Canadian journal of experimental psychology = Revue canadienne de psychologie experimentale"
      },
      "paper_id": "s2:8442c9db779fdb40f1b4e3a58b10b295fe994bb4",
      "rank": 38,
      "relevance": "该研究直接探索了近失败（near-loss）情境下的风险决策行为，类比扑克中的“接近失败”心理效应，对理解玩家在不利牌况下的决策偏差和风险管理具有启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/8442c9db779fdb40f1b4e3a58b10b295fe994bb4",
      "summary_status": "generated",
      "title": "Using the game can't stop to inform the novel behavioural state of near-loss."
    },
    {
      "abstract": "This paper presents CoupVisor, a decision-support system for the hidden-information card game Coup. It addresses two questions: what a player should do on each turn, and when a player should challenge an opponent's claim. The system is built around a single description of game events, which is shared across manual play, replay of recorded games, simulation, belief tracking, advisor recommendations, and learning-based policies. CoupVisor estimates the chance that a claim is truthful by combining how likely each role is with how many cards the claimant still holds, which corrects a case where the very first claim of a game was flagged as suspicious despite no evidence. We compare a rule-following advisor and several learned and heuristic players across many simulated games and different opponent styles. Our main finding is that the choice of reward, whether it rewards short-term gains or ultimately winning the game, decides which learning approach performs best, and that a win-oriented reward produces a policy that outperforms all baselines.",
      "evidence_level": "abstract_only",
      "first_published": "2026-08-16",
      "limitations": [
        "摘要未说明系统在真实玩家交互中的外部效度。",
        "摘要未说明除Coup外对其他卡牌游戏的泛化能力。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "奖励函数的选择是决定学习算法性能的关键：以最终获胜为导向的奖励产生的策略优于所有基线。",
        "以短期收益为导向的奖励策略表现不佳，说明长视风险判断对博弈表现的重要性。"
      ],
      "method": "构建了隐藏信息卡牌游戏Coup的决策支持系统CoupVisor，结合角色可能性与持有牌数估计声明真实性，并比较规则跟随顾问与多种学习和启发式玩家在模拟游戏中的表现。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.4304216219734653,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 12,
        "retrieval_lexical_score": 0.03529411764705882,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4145046048302371,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": ""
      },
      "paper_id": "s2:9730e4fe994e973176c6254f4edb62670f2a5ede",
      "rank": 39,
      "relevance": "尽管研究对象是Coup而非德州扑克，但同属隐藏信息博弈，该系统展示了如何在不确定信息中优化决策和风险权衡，对德州扑克策略中的信息利用与风险决策具有方法借鉴意义。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/9730e4fe994e973176c6254f4edb62670f2a5ede",
      "summary_status": "generated",
      "title": "CoupVisor: Strategy Optimization by Round and Challenge Decision Support"
    },
    {
      "abstract": "We present a metagame analysis of the competitive Pokemon Trading Card Game, machine-checked in Lean 4 over real tournament data. The headline game-theoretic results, including Nash equilibrium, replicator dynamics, and the matrix-level type-bridge computation, rely on native_decide, which trusts Lean's compiler rather than its kernel; the trust boundary is made explicit. The artifact spans approximately 31,900 lines, 87 files, and 2,627 theorems, of which roughly 200 directly verify empirical claims, with no sorry, admit, or custom axioms. Analyzing Trainer Hill data from January to February 2026 for events with at least 50 players, over 14 archetypes and their full pairwise matchup matrix, we prove a popularity paradox: the most played deck, Dragapult, with 15.5% metagame share, has only 46.7% expected win rate, while Grimmsnarl, with 5.1% share, achieves 52.7%. A machine-checked Nash equilibrium of the raw game assigns Dragapult 0% weight; exhaustive enumeration over all nonempty support subsets confirms a unique symmetric Nash equilibrium of the constant-sum symmetrization with seven-deck support. Against this equilibrium mix, Dragapult falls 40.4 permil below the game value. Single-step replicator dynamics indicate downward fitness pressure on Dragapult, upward pressure on Grimmsnarl, and strongest extinction pressure on Alakazam. A 10,000-iteration sensitivity analysis confirms qualitative stability, with core support decks appearing in more than 96% of resampled equilibria. The primary contribution is methodological: a reproducible case study showing how formal verification can turn qualitative metagame narratives into machine-checkable, re-runnable strategic science.",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-09",
      "limitations": [
        "摘要明确说明部分定理依赖native_decide，信任Lean编译器而非内核，存在可信边界。",
        "摘要未说明该结果是否适用于其他卡牌游戏或德州扑克。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "揭示了“流行度悖论”：最常使用的牌组（Dragapult）胜率仅46.7%，而较小众的Grimmsnarl胜率达52.7%。",
        "机器验证的纳什均衡将Dragapult权重定为0%，复制者动力学显示其面临向下适应压力，而Grimmsnarl有向上压力。",
        "敏感性分析表明核心支持牌组在超过96%的重采样均衡中出现，结果定性稳定。"
      ],
      "method": "使用Lean 4证明助手对宝可梦卡牌游戏的真实锦标赛数据进行元游戏分析，包括纳什均衡计算、复制者动力学和类型桥计算。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.41668817948381365,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 40,
        "retrieval_lexical_score": 0.01764705882352941,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.4046958128488137,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": ""
      },
      "paper_id": "s2:c0809ef19fcdb420ab6272562680e3f9c9c12e42",
      "rank": 40,
      "relevance": "该研究通过形式化验证将定性元游戏叙事转化为可检验的战略科学，展示了如何在竞争性博弈中应用均衡分析和动态模拟，可为德州扑克的策略分析和均衡求解提供方法论启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/c0809ef19fcdb420ab6272562680e3f9c9c12e42",
      "summary_status": "generated",
      "title": "From Rules to Nash Equilibria: A Lean 4 Case Study in Game-Theoretic Analysis of a Competitive Trading Card Game"
    },
    {
      "abstract": "In repeated games, opponents often predict what we'll do next by looking at what we have done so far. This allows us to deceive them: we can deliberately behave one way for a period of time to shape their expectations, then switch strategies to profit from the induced response. We study deception in repeated two-player normal-form games against count-based learners, whose behavior depends only on how often we have played each action in the past. We formalize deceptive and non-deceptive play, and introduce the notion of a deception bonus, the payoff gain of the best deceptive strategy over the best fixed mixed strategy. We establish structural results on deception in general-sum games. We design exact dynamic programs for optimizing against any count-based learner when the action space or opponent's memory is small, and develop approximation algorithms for settings where the opponent's memory or the time horizon is large. We also provide an approximation algorithm for learning to deceive an opponent whose count-based learning rule is unknown. To complement our algorithmic results, we show that approximating the optimal deceptive payoff against the classic Empirical Risk Minimization (ERM) learning rule is NP-hard, including obtaining any constant-factor approximation or even a $T^\\alpha$-additive approximation for any $0<\\alpha<1$. Finally, we empirically measure the deception bonus in random games with i.i.d. payoffs.",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-25",
      "limitations": [
        "摘要未说明模型对扑克这类不完全信息博弈的直接适用性。",
        "摘要未说明人类玩家是否遵循计数型学习规则。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "在一般和博弈中，针对计数型学习者，最优欺骗策略可带来额外收益（欺骗奖金）。",
        "逼近针对经验风险最小化学习规则的最优欺骗收益是NP难问题。",
        "实证测量了随机博弈中的欺骗奖金。"
      ],
      "method": "研究重复两人一般和博弈中对基于计数的学习者的欺骗策略，形式化欺骗与非欺骗行为，提出欺骗奖金概念，设计精确动态规划和近似算法，并证明逼近最优欺骗收益的困难性。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.55,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.3908275524668906,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 21,
        "retrieval_lexical_score": 0.052941176470588235,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.3846271699073413,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": ""
      },
      "paper_id": "s2:a5aa6a63fc812f6c68c69e0e6ab11433a4da3f8e",
      "rank": 41,
      "relevance": "该研究与德州扑克中的虚张声势（bluffing）和策略欺骗高度相关，揭示了在重复博弈中通过塑造对手预期来获利的结构性条件，为理解扑克中的欺骗策略提供了博弈论基础。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/a5aa6a63fc812f6c68c69e0e6ab11433a4da3f8e",
      "summary_status": "generated",
      "title": "On the Power of Deception in Repeated Games"
    },
    {
      "abstract": "Several strands of research have aimed to bridge the gap between artificial intelligence (AI) and human decision-makers in AI-assisted decision-making, where humans are the consumers of AI model predictions and the ultimate decision-makers in high-stakes applications. However, people's perception and understanding are often distorted by their cognitive biases, such as confirmation bias, anchoring bias, availability bias, to name a few. In this work, we use knowledge from the field of cognitive science to account for cognitive biases in the human-AI collaborative decision-making setting, and mitigate their negative effects on collaborative performance. To this end, we mathematically model cognitive biases and provide a general framework through which researchers and practitioners can understand the interplay between cognitive biases and human-AI accuracy. We then focus specifically on anchoring bias, a bias commonly encountered in human-AI collaboration. We implement a time-based de-anchoring strategy and conduct our first user experiment that validates its effectiveness in human-AI collaborative decision-making. With this result, we design a time allocation strategy for a resource-constrained setting that achieves optimal human-AI collaboration under some assumptions. We, then, conduct a second user experiment which shows that our time allocation strategy with explanation can effectively de-anchor the human and improve collaborative performance when the AI model has low confidence and is incorrect.",
      "evidence_level": "abstract_only",
      "first_published": "2020-10-15",
      "limitations": [
        "摘要未说明框架对其他认知偏差的适用性。",
        "摘要未说明实验环境与扑克情境的差异。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "去锚定策略和解释性时间分配策略能有效减少锚定偏差，并在AI模型低置信度且错误时提高人机协作性能。"
      ],
      "method": "基于认知科学知识对AI辅助决策中的认知偏差进行数学建模，提出统一框架，并针对锚定偏差实施基于时间的去锚定策略，设计时间分配策略并通过两个用户实验验证。",
      "metrics": {
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 2,
        "graph_foundation": 0.0,
        "graph_frontier": 0.22000000000000003,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "landscape_direction_similarity": 0.38924231635122714,
        "landscape_relevance_class": "contextual",
        "landscape_relevance_reasons": [
          "direction_embedding_contextual"
        ],
        "topic_cluster": 2,
        "topic_cluster_coherence": 0.6572789660170857,
        "topic_cluster_relation": "contextual",
        "topic_cluster_separation": 0.21417907310712875,
        "topic_cluster_title": "Cognitive Bias Amplification and Mitigation in AI-Assisted Choice"
      },
      "paper_id": "arxiv:2010.07938",
      "rank": 42,
      "relevance": "该研究直接探讨认知偏差在人机决策中的作用并给出缓解方法，对理解德州扑克中玩家的锚定效应（如率先下注影响）及如何通过调整决策时间改善判断具有启示。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "arxiv",
      "source_url": "https://arxiv.org/abs/2010.07938v2",
      "summary_status": "generated",
      "title": "Deciding Fast and Slow: The Role of Cognitive Biases in AI-assisted Decision-making"
    },
    {
      "abstract": "Reinforcement learning agents for imperfect-information card games are only as strong as the opponents they train against, and they are hard to grade, since they beat a random opponent over 99 percent of the time and only tie copies of themselves. So we build a strong, fixed, rule-based expert for Gin Rummy and use it only as a yardstick, never for training. It beats every agent we trained 70 to 99 percent of the time. Across more than a hundred runs, we isolate what makes a lightweight agent stronger. Trust region updates, a well-aimed reward, a curriculum of tougher opponents, warm starting, and keeping the best checkpoint all help, and stacking them lifts a self-play champion from about 30 to 36 percent against the expert. Several ideas did not pay off. Short-term and longer-term reward shaping, learned state embeddings, imitation and DAgger, and a live large language model opponent were each unhelpful, too slow, or too heavy to train at scale. Comparing MLP, convolutional, set-based, attention, and recurrent encoders shows that extra capacity does little to break the ceiling, suggesting the limit is information rather than network size. We add standard baselines (neural fictitious self-play and information set Monte Carlo search) and confirm the approach carries over to Leduc Hold'em, where the optimum is computable. The result is a lightweight, game-agnostic recipe that trains competitive agents without training on the expert, for any game a small model can handle, reported with robust statistics and released as a reusable package.",
      "evidence_level": "abstract_only",
      "first_published": "2026-07-07",
      "limitations": [
        "摘要未说明在真实人类玩家中的泛化表现。",
        "摘要未涉及人类决策偏差或风险管理的主题。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "信任区域更新、有明确目标的奖励、更难的对手课程、热启动和保留最佳检查点均能提升智能体强度，叠加后可使自对弈冠军对专家的胜率从30%提升到36%。",
        "短期和长期奖励塑形、学习状态嵌入、模仿学习及大语言模型对手未带来收益。",
        "增加网络容量未能突破性能上限，表明信息限制而非模型大小是关键瓶颈。"
      ],
      "method": "通过强化学习训练金拉米纸牌游戏智能体，使用固定的规则专家作为基准测试，比较不同架构和训练技巧（信任区域更新、奖励设计、课程学习、热启动、最佳检查点等）对智能体强度的影响。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 0,
        "graph_foundation": 0.0,
        "graph_frontier": 0.5531217227196821,
        "graph_reference_out_degree": 1,
        "graph_structural_role": "frontier",
        "graph_topology_community": 1,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.3891676180802582,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 34,
        "retrieval_lexical_score": 0.029411764705882353,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.3892469224543128,
        "topic_cluster": 0,
        "topic_cluster_coherence": 0.6115099447331293,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.15529618239881138,
        "topic_cluster_title": "Equilibrium Solvers and Exploitation Engines for Imperfect-Information Poker",
        "venue": ""
      },
      "paper_id": "s2:2becf3a004455e0a5ab9ec493ebf444c2edb8f64",
      "rank": 43,
      "relevance": "该研究系统分析了不完美信息博弈中训练智能体的关键因素，其关于奖励设计和信息瓶颈的发现可启发德州扑克策略研究中模型训练和评估方法，但对人类决策启示有限。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/2becf3a004455e0a5ab9ec493ebf444c2edb8f64",
      "summary_status": "generated",
      "title": "A Gold-Standard Study of What Makes a Lightweight Game-Playing Agent Strong"
    },
    {
      "abstract": "Loss aversion is considered a fundamental principle explaining risk-averse behavior in small-stakes gambling. However, this interpretation confounds loss aversion, with a preference for inaction over action. We examined whether ostensible loss aversion in gambling tasks reflects genuine aversion to losses or simply a preference for omission. Across three studies, 1,345 participants were presented with symmetric gambles under two conditions: an action/omission condition offering “bet” versus “not bet” choices and an action/action condition requiring a choice between two betting options, both stated as actions. The results consistently showed that participants in the action/action condition selected risky actions at higher rates than those in the action/omission condition. Thus, what has been attributed to apparent loss aversion in gambling contexts is fundamentally driven by preference for inaction.",
      "evidence_level": "abstract_only",
      "first_published": null,
      "limitations": [
        "摘要未说明具体实验设计细节（如赌博金额、概率结构等）以及样本的代表性。",
        "摘要未对“损失厌恶”概念本身在其他情境下的有效性进行讨论，也未说明结论的外部效度。",
        "仅依据公开摘要，未核验全文方法、实验设置与结论边界。"
      ],
      "main_findings": [
        "参与者被要求必须从两个都表述为行动的选项中选择时，选择风险行动的比例高于在行动/不行动条件下选择“赌”的比例。",
        "通常被归因于损失厌恶的赌博风险规避，实质上主要由对不作为的偏好驱动，而非对损失本身的厌恶。"
      ],
      "method": "摘要所述方法：通过三项研究，共1345名参与者，在对称赌博任务中设置行动/不行动条件（提供“赌”与“不赌”的选择）和行动/行动条件（在两个赌博选项之间选择，两者都表述为行动），比较两种条件下的风险选择比例。",
      "metrics": {
        "citation_count": 0,
        "discovered_via": "recommendation",
        "graph_bridge": 0.0,
        "graph_citation_in_degree": 0,
        "graph_community": 1,
        "graph_foundation": 0.0,
        "graph_frontier": 0.275,
        "graph_reference_out_degree": 0,
        "graph_structural_role": "frontier",
        "graph_topology_community": 0,
        "influential_citation_count": 0,
        "landscape_direction_similarity": 0.418479549467602,
        "landscape_relevance_class": "direct",
        "landscape_relevance_reasons": [
          "direction_embedding_direct"
        ],
        "reference_count": 27,
        "retrieval_lexical_score": 0.023529411764705882,
        "retrieval_relevance_class": "direct",
        "retrieval_relevance_reasons": [
          "strong_semantic_match"
        ],
        "retrieval_semantic_score": 0.41991651145869363,
        "topic_cluster": 1,
        "topic_cluster_coherence": 0.4797713319735171,
        "topic_cluster_relation": "direct",
        "topic_cluster_separation": 0.023557569639199416,
        "topic_cluster_title": "Behavioral Signatures and Bounded Rationality in Strategic Games",
        "venue": "Judgment and Decision Making"
      },
      "paper_id": "s2:771c9e534f5ea1d2f406da7d197125cc2b43da04",
      "rank": 45,
      "relevance": "本研究直接揭示了一个影响风险决策的关键认知偏差——不作为偏好，与德州扑克策略中的决策心理学相关。它提示在理解玩家决策时，不能简单将保守行为归因于损失厌恶，而需考虑对主动行动的回避倾向；对现实场景中改进风险判断和调整行为选择具有启示意义。",
      "screening_reason": "未启用 LLM 候选论文评分；使用中性占位值，候选的相对顺序由后续确定性排序决定。",
      "source": "semantic_scholar",
      "source_url": "https://www.semanticscholar.org/paper/771c9e534f5ea1d2f406da7d197125cc2b43da04",
      "summary_status": "generated",
      "title": "Omission matters: Reframing omission as action reduces apparent loss aversion"
    }
  ],
  "tags": [
    "德州扑克",
    "决策心理学",
    "风险管理",
    "认知偏差",
    "博弈论最优",
    "大语言模型"
  ],
  "title": "德州扑克里的决策课：偏差、规则与认知边界",
  "updatedAt": "2026-08-20"
}
