{
  "name": "The Arcanum AI RPG Research Index",
  "updated": "2026-09-25",
  "url": "https://arcanumrpgs.com/research/",
  "benchmark": "https://arcanumrpgs.com/benchmark/",
  "methodology": "https://arcanumrpgs.com/methodology/",
  "license": "https://creativecommons.org/licenses/by/4.0/",
  "licenseName": "CC BY 4.0",
  "note": "Every entry is verified against its primary source (arXiv API, vendor page, or Steam store page). Paper findings are paraphrased from each paper's own abstract. `axes` maps each entry to the Arcanum benchmark axes it bears on. Venues are author-stated only when venueSource says so; Semantic Scholar venues are unconfirmed. Industry entries are descriptive and unrated.",
  "groups": [
    {
      "key": "memory",
      "label": "Memory and long-horizon consistency"
    },
    {
      "key": "npc",
      "label": "Character and NPC fidelity"
    },
    {
      "key": "state",
      "label": "Game state, rules and the AI game master"
    },
    {
      "key": "agency",
      "label": "Player agency and narrative control"
    },
    {
      "key": "agents",
      "label": "AI that plays games (context)"
    },
    {
      "key": "survey",
      "label": "Surveys and field maps"
    }
  ],
  "coreRelatedWork": [
    "arxiv-2608.08160",
    "arxiv-2502.00595",
    "arxiv-2609.16614",
    "arxiv-2603.05890",
    "arxiv-2608.05170",
    "arxiv-2509.11860",
    "arxiv-2603.19313",
    "arxiv-2305.01528",
    "arxiv-2409.06949",
    "arxiv-2604.10107",
    "arxiv-2511.04962",
    "arxiv-2412.05631"
  ],
  "entries": [
    {
      "id": "arxiv-2608.08160",
      "kind": "paper",
      "group": "memory",
      "title": "Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives",
      "authors": [
        "Yingpeng Ma",
        "Jianhao Yan",
        "Bei Shi",
        "Ka Hou Kam",
        "Runnan Wang",
        "Xuebo Liu",
        "Yulong Chen",
        "Yue Zhang",
        "Derek F. Wong"
      ],
      "date": "2026-08-08",
      "venue": "Accepted by ICML 2026",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2608.08160",
      "doi": null,
      "primaryUrl": "https://arxiv.org/abs/2608.08160",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether an LLM narrator keeps its story commitments and established facts over long interactive narratives when a player pushes against them (NCP-Bench, 100 environments from movie synopses).",
      "headlineFinding": "The best model (GPT-5.2) kept its narrative intact in only 42% of runs after 20 turns; fact-conflict rates ran 40–68% across models, and high writing quality did not predict consistency.",
      "axes": [
        "memory",
        "longevity",
        "fairness"
      ],
      "limits": "Tests models with an automated adversarial player, not shipped platforms with their own memory systems.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2609.16614",
      "kind": "paper",
      "group": "memory",
      "title": "RoleBreak: Benchmarking Long-Horizon Role-Playing Robustness in Spoken Dialogue",
      "authors": [
        "Yuqi Wang",
        "Fengyuan Liu",
        "Haochen Luo",
        "Zhiqi Yu",
        "Qi Liu"
      ],
      "date": "2026-09-15",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2609.16614",
      "doi": null,
      "primaryUrl": "https://arxiv.org/abs/2609.16614",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Long-horizon role-play robustness in spoken (voice) dialogue: 310 roles, 6,688 human-verified turns.",
      "headlineFinding": "Even the strongest system hit its first persona failure after 10.4 turns on average; bigger models delayed failure but barely improved vocal emotion.",
      "axes": [
        "npc",
        "longevity"
      ],
      "limits": "Voice systems only; conversations, not games.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2606.25632",
      "kind": "paper",
      "group": "memory",
      "title": "Staying In Character: Perspective-Bounded Memory For Book-Based Role-Playing Agents",
      "authors": [
        "Xushuo Tang",
        "Junhe Zhang",
        "Zihan Yang",
        "Yifu Tang",
        "Sichao Li",
        "Longbin Lai",
        "Zhengyi Yang"
      ],
      "date": "2026-06-24",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2606.25632",
      "doi": "10.48550/arXiv.2606.25632",
      "primaryUrl": "https://arxiv.org/abs/2606.25632",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether book-based character agents stay inside what their character could know (knowledge boundaries) and keep a varied voice.",
      "headlineFinding": "A three-layer memory (episodic, visibility-tagged facts, situational personality) improved knowledge-boundary fidelity by 34.6 points over the strongest prior method.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Characters from novels, measured by question-answering, not play.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2608.05170",
      "kind": "paper",
      "group": "memory",
      "title": "DREAM: LLM-based Dynamic Role-playing via Event-Aware Memory Graph",
      "authors": [
        "Zhihao Xiao",
        "Mengting Li",
        "Xintao Wang",
        "Linfeng Li",
        "Limin Shui",
        "Mengqi Ji",
        "Borui Cai"
      ],
      "date": "2026-05-27",
      "venue": "Accepted at KDD 2026",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2608.05170",
      "doi": "10.1145/3770855.3818027",
      "primaryUrl": "https://arxiv.org/abs/2608.05170",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Temporal and causal consistency of role-playing agents (new TCM benchmark), using an event-aware memory graph (DREAM).",
      "headlineFinding": "Organising a character's experiences as a time-ordered, causally linked event graph beat strong baselines on CoSER, LifeChoice and TCM.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Established literary characters; no player-driven campaign.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2605.25693",
      "kind": "paper",
      "group": "memory",
      "title": "From Facts to Insights: A Persona-Driven Dual Memory Framework and Dataset for Role-Playing Agents",
      "authors": [
        "Rongsheng Zhang",
        "Ruofan Hu",
        "Weijie Chen",
        "Jiji Tang",
        "Junnan Ren",
        "Wanying Wu",
        "Xunuoyan Chen",
        "Tangjie Lv",
        "Tao Jin",
        "Zhou Zhao"
      ],
      "date": "2026-05-25",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2605.25693",
      "doi": "10.48550/arXiv.2605.25693",
      "primaryUrl": "https://arxiv.org/abs/2605.25693",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether memory systems interpret facts through the persona, not just store them (RoleMemo dataset, DualMem framework).",
      "headlineFinding": "Persona-agnostic summarisation yields generic answers; a 4B model with split factual/persona memory beat zero-shot DeepSeek-V3.2 frameworks on sustained persona fidelity.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Preprint; dialogue tasks, not games.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2603.19313",
      "kind": "paper",
      "group": "memory",
      "title": "Memory-Driven Role-Playing: Evaluation and Enhancement of Persona Knowledge Utilization in LLMs",
      "authors": [
        "Kai Wang",
        "Haoyang You",
        "Yang Zhang",
        "Zhongjie Wang"
      ],
      "date": "2026-03-14",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2603.19313",
      "doi": "10.48550/arXiv.2603.19313",
      "primaryUrl": "https://arxiv.org/abs/2603.19313",
      "citations": 6,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether models recall and apply their persona knowledge from dialogue context alone (MREval: Anchoring, Recalling, Bounding, Enacting; MRBench).",
      "headlineFinding": "A structured retrieval prompt let small models (Qwen3-8B) match much larger closed models, and memory gains carried through to response quality.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Single-character dialogue; 12 models, no platforms.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2603.05890",
      "kind": "paper",
      "group": "memory",
      "title": "Lost in Stories: Consistency Bugs in Long Story Generation by LLMs",
      "authors": [
        "Junjie Li",
        "Xinrui Guo",
        "Yuhao Wu",
        "Roy Ka-Wei Lee",
        "Hongzhi Li",
        "Yutao Xie"
      ],
      "date": "2026-03-06",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2603.05890",
      "doi": "10.48550/arXiv.2603.05890",
      "primaryUrl": "https://arxiv.org/abs/2603.05890",
      "citations": 8,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Consistency errors in long-form story generation (ConStory-Bench: 2,000 prompts, 19 error subtypes).",
      "headlineFinding": "Contradictions cluster in factual and temporal details and tend to appear around the middle of long narratives.",
      "axes": [
        "memory",
        "longevity"
      ],
      "limits": "Non-interactive story generation.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2509.11860",
      "kind": "paper",
      "group": "memory",
      "title": "MOOM: Maintenance, Organization and Optimization of Memory in Ultra-Long Role-Playing Dialogues",
      "authors": [
        "Weishu Chen",
        "Jinyi Tang",
        "Zhouhui Hou",
        "Shihao Han",
        "Mingjie Zhan",
        "Zhiyuan Huang",
        "Delong Liu",
        "Jiawei Guo",
        "Zhicheng Zhao",
        "Fei Su"
      ],
      "date": "2025-09-15",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2509.11860",
      "doi": "10.48550/arXiv.2509.11860",
      "primaryUrl": "https://arxiv.org/abs/2509.11860",
      "citations": 6,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Memory extraction for ultra-long role-play dialogues (Chinese ZH-4O dataset, ~600 turns per dialogue).",
      "headlineFinding": "A dual-branch memory (plot conflicts + user profile) with a forgetting mechanism kept memory size controlled while beating prior extraction methods.",
      "axes": [
        "memory",
        "longevity"
      ],
      "limits": "Chinese-language dialogue data; memory extraction measured, not play experience.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2506.13356",
      "kind": "paper",
      "group": "memory",
      "title": "StoryBench: A Dynamic Benchmark for Evaluating Long-Term Memory with Multi Turns",
      "authors": [
        "Luanbo Wan",
        "Weizhi Ma"
      ],
      "date": "2025-06-16",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2506.13356",
      "doi": "10.48550/arXiv.2506.13356",
      "primaryUrl": "https://arxiv.org/abs/2506.13356",
      "citations": 15,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Long-term memory measured through branching interactive-fiction games (StoryBench).",
      "headlineFinding": "Interactive fiction works as a test bed for long-term memory: choices cascade across turns, and models must trace back and revise earlier decisions.",
      "axes": [
        "memory",
        "agency"
      ],
      "limits": "Measures models as players, not as narrators.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2511.10277",
      "kind": "paper",
      "group": "memory",
      "title": "Fixed-Persona SLMs with Modular Memory: Scalable NPC Dialogue on Consumer Hardware",
      "authors": [
        "Martin Braas",
        "Lukas Esterle"
      ],
      "date": "2025-11-13",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2511.10277",
      "doi": "10.48550/arXiv.2511.10277",
      "primaryUrl": "https://arxiv.org/abs/2511.10277",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "NPC dialogue with small fine-tuned models plus swappable memory modules on consumer hardware.",
      "headlineFinding": "Persona-tuned small models with runtime-swappable memory kept character context without reloading, on consumer GPUs.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Engineering study with small models; no player evaluation of fun.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2404.13501",
      "kind": "paper",
      "group": "survey",
      "title": "A Survey on the Memory Mechanism of Large Language Model based Agents",
      "authors": [
        "Zeyu Zhang",
        "Xiaohe Bo",
        "Chen Ma",
        "Rui Li",
        "Xu Chen",
        "Quanyu Dai",
        "Jieming Zhu",
        "Zhenhua Dong",
        "Ji-Rong Wen"
      ],
      "date": "2024-04-21",
      "venue": "ACM Trans. Inf. Syst.",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2404.13501",
      "doi": "10.1145/3748302",
      "primaryUrl": "https://arxiv.org/abs/2404.13501",
      "citations": 796,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Survey of memory mechanisms in LLM-based agents: design and evaluation of memory modules.",
      "headlineFinding": "The standard reference for how agent memory is built and evaluated.",
      "axes": [
        "memory"
      ],
      "limits": "General agents, not games.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2305.13304",
      "kind": "paper",
      "group": "memory",
      "title": "RecurrentGPT: Interactive Generation of (Arbitrarily) Long Text",
      "authors": [
        "Wangchunshu Zhou",
        "Yuchen Eleanor Jiang",
        "Peng Cui",
        "Tiannan Wang",
        "Zhenxin Xiao",
        "Yifan Hou",
        "Ryan Cotterell",
        "Mrinmaya Sachan"
      ],
      "date": "2023-05-22",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2305.13304",
      "doi": "10.48550/arXiv.2305.13304",
      "primaryUrl": "https://arxiv.org/abs/2305.13304",
      "citations": 103,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Generating arbitrarily long text by simulating long/short-term memory in natural language (RecurrentGPT).",
      "headlineFinding": "Human-readable, editable memories let an LLM write past its context window; demonstrated as personalised interactive fiction.",
      "axes": [
        "memory",
        "longevity"
      ],
      "limits": "2023 system; predates today's long-context models.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2606.05553",
      "kind": "paper",
      "group": "npc",
      "title": "ArcANE: Do Role-Playing Language Agents Stay in Character at the Right Time?",
      "authors": [
        "Woojung Song",
        "Nalim Kim",
        "Sangjun Song",
        "Chaewon Heo",
        "Jongwon Lim",
        "Yohan Jo"
      ],
      "date": "2026-06-04",
      "venue": "Accepted at EMNLP 2026 (Main)",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2606.05553",
      "doi": "10.48550/arXiv.2606.05553",
      "primaryUrl": "https://arxiv.org/abs/2606.05553",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether role-playing agents follow a character's development over a story rather than a fixed persona (ArcANE).",
      "headlineFinding": "Giving the model the character's arc up to the current chapter beat every other context strategy by 2.2–8.4 points in all six models.",
      "axes": [
        "npc",
        "memory"
      ],
      "limits": "Novel characters, question-style scenarios.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2511.04962",
      "kind": "paper",
      "group": "npc",
      "title": "Too Good to be Bad: On the Failure of LLMs to Role-Play Villains",
      "authors": [
        "Zihao Yi",
        "Qingxuan Jiang",
        "Ruotian Ma",
        "Xingyu Chen",
        "Qu Yang",
        "Mengru Wang",
        "Fanghua Ye",
        "Ying Shen",
        "Zhaopeng Tu",
        "Xiaolong Li",
        " Linus"
      ],
      "date": "2025-11-07",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2511.04962",
      "doi": "10.48550/arXiv.2511.04962",
      "primaryUrl": "https://arxiv.org/abs/2511.04962",
      "citations": 11,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "How faithfully models play characters across a four-level moral scale, from paragons to villains (Moral RolePlay).",
      "headlineFinding": "Fidelity falls steadily as characters become less moral; safety-aligned models swap nuanced malice for surface aggression, and chatbot skill does not predict villain skill.",
      "axes": [
        "npc",
        "signature"
      ],
      "limits": "Scripted scenes; measures the model, not a platform's prompt design.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2510.13586",
      "kind": "paper",
      "group": "npc",
      "title": "Deflanderization for Game Dialogue: Balancing Character Authenticity with Task Execution in LLM-based NPCs",
      "authors": [
        "Pasin Buakhaw",
        "Kun Kerdthaisong",
        "Phuree Phenhiran",
        "Pitikorn Khlaisamniang",
        "Supasate Vorathammathorn",
        "Piyalitt Ittichaiwong",
        "Nutchanon Yongsatianchot"
      ],
      "date": "2025-10-15",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2510.13586",
      "doi": "10.48550/arXiv.2510.13586",
      "primaryUrl": "https://arxiv.org/abs/2510.13586",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "NPCs that must both stay in character and complete game tasks (Commonsense Persona-Grounded Dialogue Challenge 2025).",
      "headlineFinding": "Prompting that suppresses excessive role-play (\"Deflanderization\") improved task fidelity; competition entry placed 2nd on two tasks.",
      "axes": [
        "npc",
        "depth"
      ],
      "limits": "Competition report, short.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2510.25820",
      "kind": "paper",
      "group": "npc",
      "title": "Symbolically Scaffolded Play: Designing Role-Sensitive Prompts for Generative NPC Dialogue",
      "authors": [
        "Vanessa Figueiredo",
        "David Elumeze"
      ],
      "date": "2025-10-29",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2510.25820",
      "doi": "10.48550/arXiv.2510.25820",
      "primaryUrl": "https://arxiv.org/abs/2510.25820",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether tighter prompt constraints improve player experience with generative NPCs (voice detective game, N=10 study + LLM judge).",
      "headlineFinding": "Scaffolding helped the quest-giver NPC but made suspect NPCs less believable — tighter constraints do not automatically make play better.",
      "axes": [
        "npc",
        "agency"
      ],
      "limits": "Very small user study.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2412.11912",
      "kind": "paper",
      "group": "npc",
      "title": "CharacterBench: Benchmarking Character Customization of Large Language Models",
      "authors": [
        "Jinfeng Zhou",
        "Yongkang Huang",
        "Bosi Wen",
        "Guanqun Bi",
        "Yuxuan Chen",
        "Pei Ke",
        "Zhuang Chen",
        "Xiyao Xiao",
        "Libiao Peng",
        "Kuntian Tang",
        "Rongsheng Zhang",
        "Le Zhang",
        "Tangjie Lv",
        "Zhipeng Hu",
        "Hongning Wang",
        "Minlie Huang"
      ],
      "date": "2024-12-16",
      "venue": "AAAI 2025",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2412.11912",
      "doi": "10.48550/arXiv.2412.11912",
      "primaryUrl": "https://arxiv.org/abs/2412.11912",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Character customisation ability of LLMs across 11 dimensions (CharacterBench, 22,859 human-annotated samples, 3,956 characters).",
      "headlineFinding": "Largest bilingual character benchmark; its trained judge beat GPT-4 as an evaluator.",
      "axes": [
        "npc"
      ],
      "limits": "Dialogue snapshots, not sustained play.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2412.05631",
      "kind": "paper",
      "group": "npc",
      "title": "CharacterBox: Evaluating the Role-Playing Capabilities of LLMs in Text-Based Virtual Worlds",
      "authors": [
        "Lei Wang",
        "Jianxun Lian",
        "Yi Huang",
        "Yanqi Dai",
        "Haoxuan Li",
        "Xu Chen",
        "Xing Xie",
        "Ji-Rong Wen"
      ],
      "date": "2024-12-07",
      "venue": "North American Chapter of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2412.05631",
      "doi": "10.48550/arXiv.2412.05631",
      "primaryUrl": "https://arxiv.org/abs/2412.05631",
      "citations": 48,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Role-play in simulated text worlds with a character agent and a narrator agent (CharacterBox).",
      "headlineFinding": "Behaviour trajectories in a sandbox give a deeper read of role-play than Q&A snapshots.",
      "axes": [
        "npc",
        "agency"
      ],
      "limits": "Simulated users; no human players.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2310.00746",
      "kind": "paper",
      "group": "npc",
      "title": "RoleLLM: Benchmarking, Eliciting, and Enhancing Role-Playing Abilities of Large Language Models",
      "authors": [
        "Zekun Moore Wang",
        "Zhongyuan Peng",
        "Haoran Que",
        "Jiaheng Liu",
        "Wangchunshu Zhou",
        "Yuhan Wu",
        "Hongcheng Guo",
        "Ruitong Gan",
        "Zehao Ni",
        "Jian Yang",
        "Man Zhang",
        "Zhaoxiang Zhang",
        "Wanli Ouyang",
        "Ke Xu",
        "Stephen W. Huang",
        "Jie Fu",
        "Junran Peng"
      ],
      "date": "2023-10-01",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2310.00746",
      "doi": "10.48550/arXiv.2310.00746",
      "primaryUrl": "https://arxiv.org/abs/2310.00746",
      "citations": 257,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Benchmarking and improving role-playing ability (RoleBench, 168,093 samples, 100 roles).",
      "headlineFinding": "Role-conditioned tuning let open models approach GPT-4 role-play.",
      "axes": [
        "npc"
      ],
      "limits": "2023 models; speaking-style focus.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2310.17976",
      "kind": "paper",
      "group": "npc",
      "title": "InCharacter: Evaluating Personality Fidelity in Role-Playing Agents through Psychological Interviews",
      "authors": [
        "Xintao Wang",
        "Yunze Xiao",
        "Jen-tse Huang",
        "Siyu Yuan",
        "Rui Xu",
        "Haoran Guo",
        "Quan Tu",
        "Yaying Fei",
        "Ziang Leng",
        "Wei Wang",
        "Jiangjie Chen",
        "Cheng Li",
        "Yanghua Xiao"
      ],
      "date": "2023-10-27",
      "venue": "ACL 2024",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2310.17976",
      "doi": "10.18653/v1/2024.acl-long.102",
      "primaryUrl": "https://arxiv.org/abs/2310.17976",
      "citations": 240,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Personality fidelity of role-playing agents measured by psychological interviews (InCharacter, 32 characters, 14 scales).",
      "headlineFinding": "State-of-the-art agents matched human-perceived character personalities with up to 80.7% accuracy.",
      "axes": [
        "npc"
      ],
      "limits": "Personality tests, not gameplay.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2401.01275",
      "kind": "paper",
      "group": "npc",
      "title": "CharacterEval: A Chinese Benchmark for Role-Playing Conversational Agent Evaluation",
      "authors": [
        "Quan Tu",
        "Shilong Fan",
        "Zihang Tian",
        "Rui Yan"
      ],
      "date": "2024-01-02",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2401.01275",
      "doi": "10.48550/arXiv.2401.01275",
      "primaryUrl": "https://arxiv.org/abs/2401.01275",
      "citations": 184,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Chinese role-playing conversational agents across 13 metrics (CharacterEval, 77 characters).",
      "headlineFinding": "Chinese LLMs outperformed GPT-4 in Chinese role-play conversation.",
      "axes": [
        "npc"
      ],
      "limits": "Chinese-language only.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2503.08193",
      "kind": "paper",
      "group": "npc",
      "title": "Guess What I am Thinking: A Benchmark for Inner Thought Reasoning of Role-Playing Language Agents",
      "authors": [
        "Rui Xu",
        "MingYu Wang",
        "XinTao Wang",
        "Dakuan Lu",
        "Xiaoyu Tan",
        "Wei Chu",
        "Yinghui Xu"
      ],
      "date": "2025-03-11",
      "venue": "Conference on Empirical Methods in Natural Language Processing",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2503.08193",
      "doi": "10.48550/arXiv.2503.08193",
      "primaryUrl": "https://arxiv.org/abs/2503.08193",
      "citations": 11,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether role-playing agents can generate a character's inner thoughts (RoleThink).",
      "headlineFinding": "Retrieving memories and predicting reactions before answering improved inner-thought generation.",
      "axes": [
        "npc"
      ],
      "limits": "Literary characters.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2502.09082",
      "kind": "paper",
      "group": "npc",
      "title": "CoSER: A Comprehensive Literary Dataset and Framework for Training and Evaluating LLM Role-Playing and Persona Simulation",
      "authors": [
        "Xintao Wang",
        "Heng Wang",
        "Yifei Zhang",
        "Xinfeng Yuan",
        "Rui Xu",
        "Jen-tse Huang",
        "Siyu Yuan",
        "Haoran Guo",
        "Jiangjie Chen",
        "Shuchang Zhou",
        "Wei Wang",
        "Yanghua Xiao"
      ],
      "date": "2025-02-13",
      "venue": "Accepted by ICML 2025",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2502.09082",
      "doi": null,
      "primaryUrl": "https://arxiv.org/abs/2502.09082",
      "citations": 24,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Training and evaluating role-play of established characters from 771 books (CoSER, 17,966 characters).",
      "headlineFinding": "Its open 70B model matched or beat GPT-4o on its own and three existing benchmarks.",
      "axes": [
        "npc"
      ],
      "limits": "Book scenes, not games.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2507.20352",
      "kind": "paper",
      "group": "npc",
      "title": "RMTBench: Benchmarking LLMs Through Multi-Turn User-Centric Role-Playing",
      "authors": [
        "Hao Xiang",
        "Tianyi Tang",
        "Yang Su",
        "Bowen Yu",
        "An Yang",
        "Fei Huang",
        "Yichang Zhang",
        "Yaojie Lu",
        "Hongyu Lin",
        "Xianpei Han",
        "Jingren Zhou",
        "Junyang Lin",
        "Le Sun"
      ],
      "date": "2025-07-27",
      "venue": "Conference on Empirical Methods in Natural Language Processing",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2507.20352",
      "doi": "10.48550/arXiv.2507.20352",
      "primaryUrl": "https://arxiv.org/abs/2507.20352",
      "citations": 10,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "User-centric multi-turn role-play (RMTBench, 80 characters, 8,000+ rounds).",
      "headlineFinding": "Built dialogues around what the user wants rather than the character sheet, closer to real use.",
      "axes": [
        "npc",
        "agency"
      ],
      "limits": "Simulated users, LLM scoring.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2409.06820",
      "kind": "paper",
      "group": "npc",
      "title": "PingPong: A Benchmark for Role-Playing Language Models with User Emulation and Multi-Model Evaluation",
      "authors": [
        "Ilya Gusev"
      ],
      "date": "2024-09-10",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2409.06820",
      "doi": "10.48550/arXiv.2409.06820",
      "primaryUrl": "https://arxiv.org/abs/2409.06820",
      "citations": 11,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Role-play quality via simulated users and a judge ensemble: character consistency, entertainment, fluency (PingPong, 40+ models).",
      "headlineFinding": "Automated judgments correlated strongly with human annotations.",
      "axes": [
        "npc",
        "signature"
      ],
      "limits": "LLM-judged; English and Russian.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2405.18027",
      "kind": "paper",
      "group": "npc",
      "title": "TimeChara: Evaluating Point-in-Time Character Hallucination of Role-Playing Large Language Models",
      "authors": [
        "Jaewoo Ahn",
        "Taehyun Lee",
        "Junyoung Lim",
        "Jin-Hwa Kim",
        "Sangdoo Yun",
        "Hwaran Lee",
        "Gunhee Kim"
      ],
      "date": "2024-05-28",
      "venue": "ACL 2024 Findings",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2405.18027",
      "doi": "10.48550/arXiv.2405.18027",
      "primaryUrl": "https://arxiv.org/abs/2405.18027",
      "citations": 30,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Point-in-time character hallucination: characters knowing things they should not yet know (TimeChara, 10,895 instances).",
      "headlineFinding": "Significant hallucination even in GPT-4o.",
      "axes": [
        "npc",
        "memory"
      ],
      "limits": "Fandom characters, question format.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2609.01352",
      "kind": "paper",
      "group": "npc",
      "title": "CHARM: Character Hallucination for Multicultural Role Play Benchmark",
      "authors": [
        "Sunkyung Han",
        "Nahyeon Park",
        "Gaeun Seo",
        "Seunghyun Yoon",
        "JinYeong Bak"
      ],
      "date": "2026-09-01",
      "venue": "Accepted to Findings of EMNLP 2026",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2609.01352",
      "doi": null,
      "primaryUrl": "https://arxiv.org/abs/2609.01352",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether characters respect knowledge boundaries across five cultural regions (CHARM, 40 characters).",
      "headlineFinding": "Hallucination comes mostly from compliance failures: models recognise a question is out of character and answer it anyway.",
      "axes": [
        "npc"
      ],
      "limits": "Multiple-choice probes.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2508.19288",
      "kind": "paper",
      "group": "npc",
      "title": "Tricking LLM-Based NPCs into Spilling Secrets",
      "authors": [
        "Kyohei Shiomi",
        "Zhuotao Lian",
        "Toru Nakanishi",
        "Teruaki Kitasuka"
      ],
      "date": "2025-08-25",
      "venue": "Provable Security",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2508.19288",
      "doi": "10.48550/arXiv.2508.19288",
      "primaryUrl": "https://arxiv.org/abs/2508.19288",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Whether prompt injection can make LLM NPCs reveal secrets they are meant to keep.",
      "headlineFinding": "Adversarial prompts can make LLM-based NPCs spill hidden background secrets — a security issue for game design.",
      "axes": [
        "npc",
        "fairness"
      ],
      "limits": "Short study.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2508.17825",
      "kind": "paper",
      "group": "npc",
      "title": "FAIRGAMER: Evaluating Social Biases in LLM-Based Video Game NPCs",
      "authors": [
        "Bingkang Shi",
        "Jen-tse Huang",
        "Long Luo",
        "Tianyu Zong",
        "Hongzhu Yi",
        "Yuanxiang Wang",
        "Songlin Hu",
        "Xiaodan Zhang",
        "Zhongjiang Yao"
      ],
      "date": "2025-08-25",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2508.17825",
      "doi": "10.18653/v1/2026.acl-long.2015",
      "primaryUrl": "https://arxiv.org/abs/2508.17825",
      "citations": 2,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Social bias in LLM NPC decisions across trade, cooperation and competition (FairGamer).",
      "headlineFinding": "All seven frontier models showed biased decisions, and larger models showed more bias.",
      "axes": [
        "npc",
        "fairness"
      ],
      "limits": "Decision tasks, not dialogue quality.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2604.10107",
      "kind": "paper",
      "group": "npc",
      "title": "The Double-Edged Sword of Open-Ended Interaction: How LLM-Driven NPCs Affect Players' Cognitive Load and Gaming Experience",
      "authors": [
        "Ting-Chen Hsu",
        "Wenran Chen",
        "Jiangxu Lin",
        "Fei Qin",
        "Zheyuan Zhang"
      ],
      "date": "2026-04-11",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2604.10107",
      "doi": "10.48550/arXiv.2604.10107",
      "primaryUrl": "https://arxiv.org/abs/2604.10107",
      "citations": 4,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "How LLM NPCs affect players' cognitive load and experience versus scripted NPCs (randomised study, N=130).",
      "headlineFinding": "LLM NPCs significantly raised cognitive load and did not significantly improve overall experience; autonomy rose while usability and trust fell.",
      "axes": [
        "agency",
        "npc",
        "signature"
      ],
      "limits": "One research prototype game.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2507.10469",
      "kind": "paper",
      "group": "npc",
      "title": "An Empirical Evaluation of AI-Powered Non-Player Characters' Perceived Realism and Performance in Virtual Reality Environments",
      "authors": [
        "Mikko Korkiakoski",
        "Saeid Sheikhi",
        "Jesper Nyman",
        "Jussi Saariniemi",
        "Kalle Tapio",
        "Panos Kostakos"
      ],
      "date": "2025-07-14",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2507.10469",
      "doi": "10.48550/arXiv.2507.10469",
      "primaryUrl": "https://arxiv.org/abs/2507.10469",
      "citations": 6,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Perceived realism and latency of GPT-4-driven NPCs in a VR interrogation game (N=18).",
      "headlineFinding": "Believability 6.67/10 and a 7-second average response cycle that grew with conversation context.",
      "axes": [
        "npc"
      ],
      "limits": "Small sample; VR voice setting.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2502.00595",
      "kind": "paper",
      "group": "state",
      "title": "RPGBENCH: Evaluating Large Language Models as Role-Playing Game Engines",
      "authors": [
        "Pengfei Yu",
        "Dongming Shen",
        "Silin Meng",
        "Jaewon Lee",
        "Weisu Yin",
        "Andrea Yaoyun Cui",
        "Zhenlin Xu",
        "Yi Zhu",
        "Xingjian Shi",
        "Mu Li",
        "Alex Smola"
      ],
      "date": "2025-02-01",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2502.00595",
      "doi": "10.48550/arXiv.2502.00595",
      "primaryUrl": "https://arxiv.org/abs/2502.00595",
      "citations": 13,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "LLMs as text RPG engines: creating a valid game (event-state representation) and simulating play while updating state and enforcing rules (RPGBench).",
      "headlineFinding": "State-of-the-art LLMs produce engaging stories but often fail to implement consistent, verifiable game mechanics, especially in long or complex scenarios.",
      "axes": [
        "depth",
        "fairness",
        "memory"
      ],
      "limits": "Models tested as engines, not shipped platforms with external state.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2605.24719",
      "kind": "paper",
      "group": "state",
      "title": "World-State Transformations for Neuro-symbolic Interactive Storytelling",
      "authors": [
        "Santiago Góngora",
        "Luis Chiruzzo",
        "Gonzalo Méndez",
        "Pablo Gervás"
      ],
      "date": "2026-05-23",
      "venue": "To be presented at the 17th International Conference on Computational Creativity (ICCC'26)",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2605.24719",
      "doi": "10.48550/arXiv.2605.24719",
      "primaryUrl": "https://arxiv.org/abs/2605.24719",
      "citations": 2,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Neuro-symbolic interactive storytelling where the LLM triggers pre-programmed world-state changes.",
      "headlineFinding": "World-state transformations kept the world consistent while still encouraging creative player input.",
      "axes": [
        "depth",
        "fairness",
        "agency"
      ],
      "limits": "Eight participants; exploratory.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2606.13348",
      "kind": "paper",
      "group": "state",
      "title": "IVIE: A Neuro-symbolic Approach to Incremental and Validated Generation of Interactive Fiction Worlds",
      "authors": [
        "Micaela Vaucher",
        "Santiago Silveira",
        "Santiago Góngora",
        "Luis Chiruzzo"
      ],
      "date": "2026-06-11",
      "venue": "To appear in the Proceedings of the 16th International Conference on Computational Creativity (ICCC'26), June 2026",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2606.13348",
      "doi": "10.48550/arXiv.2606.13348",
      "primaryUrl": "https://arxiv.org/abs/2606.13348",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Generating complete, playable interactive-fiction worlds with symbolic validation (IVIE).",
      "headlineFinding": "Symbolic validation grounded the LLM without removing creative freedom, though model inconsistencies occasionally bypassed puzzles.",
      "axes": [
        "depth",
        "fairness"
      ],
      "limits": "Generation of IF worlds, not live campaigns.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2504.07304",
      "kind": "paper",
      "group": "state",
      "title": "PAYADOR: A Minimalist Approach to Grounding Language Models on Structured Data for Interactive Storytelling and Role-playing Games",
      "authors": [
        "Santiago Góngora",
        "Luis Chiruzzo",
        "Gonzalo Méndez",
        "Pablo Gervás"
      ],
      "date": "2025-04-09",
      "venue": "Proceedings of the Fifteenth International Conference on Computational Creativity",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2504.07304",
      "doi": "10.48550/arXiv.2504.07304",
      "primaryUrl": "https://arxiv.org/abs/2504.07304",
      "citations": 2,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Grounding an LLM on a minimal world representation by predicting action outcomes (PAYADOR).",
      "headlineFinding": "Predicting outcomes instead of mapping input to fixed actions preserves player freedom in RPG-style play.",
      "axes": [
        "depth",
        "fairness",
        "agency"
      ],
      "limits": "Minimal prototype.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2409.06949",
      "kind": "paper",
      "group": "state",
      "title": "You Have Thirteen Hours in Which to Solve the Labyrinth: Enhancing AI Game Masters with Function Calling",
      "authors": [
        "Jaewoo Song",
        "Andrew Zhu",
        "Chris Callison-Burch"
      ],
      "date": "2024-09-11",
      "venue": "Wordplay Workshop @ ACL 2024",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2409.06949",
      "doi": "10.48550/arXiv.2409.06949",
      "primaryUrl": "https://arxiv.org/abs/2409.06949",
      "citations": 6,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "AI game masters using function calling for game-specific controls in a tabletop RPG (Labyrinth: The Adventure Game).",
      "headlineFinding": "Function calling improved both narrative quality and state-update consistency, by human evaluation and unit tests.",
      "axes": [
        "depth",
        "fairness",
        "memory"
      ],
      "limits": "One game system; workshop paper.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2305.01528",
      "kind": "paper",
      "group": "state",
      "title": "FIREBALL: A Dataset of Dungeons and Dragons Actual-Play with Structured Game State Information",
      "authors": [
        "Andrew Zhu",
        "Karmanya Aggarwal",
        "Alexander Feng",
        "Lara J. Martin",
        "Chris Callison-Burch"
      ],
      "date": "2023-05-02",
      "venue": "Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2305.01528",
      "doi": "10.18653/v1/2023.acl-long.229",
      "primaryUrl": "https://arxiv.org/abs/2305.01528",
      "citations": 24,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "D&D actual play with true game state from ~25,000 Discord sessions using the Avrae bot (FIREBALL).",
      "headlineFinding": "Giving models real game-state information improved generated turns; models can also produce executable game commands.",
      "axes": [
        "depth",
        "fairness"
      ],
      "limits": "Human play logs; model generation evaluated, not full GM play.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2308.07540",
      "kind": "paper",
      "group": "state",
      "title": "CALYPSO: LLMs as Dungeon Masters' Assistants",
      "authors": [
        "Andrew Zhu",
        "Lara J. Martin",
        "Andrew Head",
        "Chris Callison-Burch"
      ],
      "date": "2023-08-15",
      "venue": "AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment (AIIDE) 2023",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2308.07540",
      "doi": "10.1609/aiide.v19i1.27534",
      "primaryUrl": "https://arxiv.org/abs/2308.07540",
      "citations": 44,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "LLM tools that assist a human Dungeon Master (CALYPSO).",
      "headlineFinding": "DMs found the output good enough to read to players and used it for ideas while keeping creative control.",
      "axes": [
        "depth",
        "signature"
      ],
      "limits": "Assists a human DM rather than replacing one.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2505.22809",
      "kind": "paper",
      "group": "state",
      "title": "First Steps Towards Overhearing LLM Agents: A Case Study With Dungeons & Dragons Gameplay",
      "authors": [
        "Andrew Zhu",
        "Evan Osgood",
        "Chris Callison-Burch"
      ],
      "date": "2025-05-28",
      "venue": "COLM 2025 Workshop on AI Agents",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2505.22809",
      "doi": "10.48550/arXiv.2505.22809",
      "primaryUrl": "https://arxiv.org/abs/2505.22809",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "\"Overhearing\" agents that listen to a D&D table and help the DM in the background.",
      "headlineFinding": "Some large audio-language models can perform background assistance from implicit audio cues.",
      "axes": [
        "depth"
      ],
      "limits": "Workshop paper; assistive, not autonomous.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2210.07109",
      "kind": "paper",
      "group": "state",
      "title": "Dungeons and Dragons as a Dialog Challenge for Artificial Intelligence",
      "authors": [
        "Chris Callison-Burch",
        "Gaurav Singh Tomar",
        "Lara J. Martin",
        "Daphne Ippolito",
        "Suma Bailis",
        "David Reitter"
      ],
      "date": "2022-10-13",
      "venue": "Conference on Empirical Methods in Natural Language Processing (EMNLP), pp",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2210.07109",
      "doi": "10.18653/v1/2022.emnlp-main.637",
      "primaryUrl": "https://arxiv.org/abs/2210.07109",
      "citations": 65,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "D&D as a dialogue challenge: generating the next turn and predicting game state (~900 games, 800,000 turns).",
      "headlineFinding": "Tracking game state measurably improves the model's turns; frames D&D as a benchmark problem for AI.",
      "axes": [
        "depth",
        "npc"
      ],
      "limits": "Pre-ChatGPT models.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2304.01860",
      "kind": "paper",
      "group": "state",
      "title": "Rolling the Dice: Imagining Generative AI as a Dungeons & Dragons Storytelling Companion",
      "authors": [
        "Jose Ma. Santiago",
        "Richard Lance Parayno",
        "Jordan Aiko Deja",
        "Briane Paul V. Samson"
      ],
      "date": "2023-04-04",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2304.01860",
      "doi": "10.48550/arXiv.2304.01860",
      "primaryUrl": "https://arxiv.org/abs/2304.01860",
      "citations": 10,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Design vision for generative AI as a D&D storytelling companion.",
      "headlineFinding": "Proposes design guidelines and flags immersion and cognitive-load questions.",
      "axes": [
        "agency",
        "signature"
      ],
      "limits": "Position paper, no experiment.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2604.25482",
      "kind": "paper",
      "group": "state",
      "title": "From World-Gen to Quest-Line: A Dependency-Driven Prompt Pipeline for Coherent RPG Generation",
      "authors": [
        "Dominik Borawski",
        "Marta Szulc",
        "Robert Chudy",
        "Małgorzata Giedrowicz",
        "Piotr Mironowicz"
      ],
      "date": "2026-04-28",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2604.25482",
      "doi": "10.48550/arXiv.2604.25482",
      "primaryUrl": "https://arxiv.org/abs/2604.25482",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "A staged, schema-enforced prompt pipeline for generating RPG worlds, NPCs and quests.",
      "headlineFinding": "Structured JSON hand-offs between stages reduced drift and kept content valid as complexity grew.",
      "axes": [
        "depth",
        "signature"
      ],
      "limits": "Qualitative evaluation.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2505.03547",
      "kind": "paper",
      "group": "state",
      "title": "STORY2GAME: Generating (Almost) Everything in an Interactive Fiction Game",
      "authors": [
        "Eric Zhou",
        "Shreyas Basavatia",
        "Moontashir Siam",
        "Zexin Chen",
        "Mark O. Riedl"
      ],
      "date": "2025-05-06",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2505.03547",
      "doi": "10.48550/arXiv.2505.03547",
      "primaryUrl": "https://arxiv.org/abs/2505.03547",
      "citations": 6,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Generating interactive-fiction games including the action code, with new actions created on demand (STORY2GAME).",
      "headlineFinding": "Generating action preconditions and effects lets stories stay open-ended yet grounded in tracked game state.",
      "axes": [
        "depth",
        "agency"
      ],
      "limits": "Generated games, not live play.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2606.16014",
      "kind": "paper",
      "group": "state",
      "title": "Orchestrated Reality: From Role-Play to Living, Playable Game Worlds -- LLM-Driven World Simulation as a Parameterized-Action POMDP",
      "authors": [
        "Yuhang Huang",
        "Chenmiao Li",
        "Chaowei Fang"
      ],
      "date": "2026-06-14",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2606.16014",
      "doi": "10.48550/arXiv.2606.16014",
      "primaryUrl": "https://arxiv.org/abs/2606.16014",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Architecture for LLM-run game worlds where the world is a canonical object owned by an orchestrator (work in progress).",
      "headlineFinding": "Argues that deployed systems let the narrator assert state in free prose \"without any validated representation\", so a fully autonomous engine remains infeasible.",
      "axes": [
        "depth",
        "memory",
        "fairness"
      ],
      "limits": "Work in progress; no evaluation yet.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2312.17653",
      "kind": "paper",
      "group": "state",
      "title": "LARP: Language-Agent Role Play for Open-World Games",
      "authors": [
        "Ming Yan",
        "Ruihao Li",
        "Hao Zhang",
        "Hao Wang",
        "Zhilan Yang",
        "Ji Yan"
      ],
      "date": "2023-12-24",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2312.17653",
      "doi": "10.48550/arXiv.2312.17653",
      "primaryUrl": "https://arxiv.org/abs/2312.17653",
      "citations": 27,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Cognitive architecture for role-playing agents in open-world games (LARP).",
      "headlineFinding": "Combines memory processing, learnable action space and personality alignment for open-world NPCs.",
      "axes": [
        "memory",
        "npc"
      ],
      "limits": "Framework paper.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2308.13548",
      "kind": "paper",
      "group": "agents",
      "title": "Towards a Holodeck-style Simulation Game",
      "authors": [
        "Ahad Shams",
        "Douglas Summers-Stay",
        "Arpan Tripathi",
        "Vsevolod Metelsky",
        "Alexandros Titonis",
        "Karan Malhotra"
      ],
      "date": "2023-08-22",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2308.13548",
      "doi": "10.48550/arXiv.2308.13548",
      "primaryUrl": "https://arxiv.org/abs/2308.13548",
      "citations": 3,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "A Holodeck-style simulation game built on generative agents (Infinitia).",
      "headlineFinding": "Applies the Generative Agents idea to a playable, multiplayer Unity sandbox.",
      "axes": [
        "signature"
      ],
      "limits": "System description.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2404.17027",
      "kind": "paper",
      "group": "agency",
      "title": "Player-Driven Emergence in LLM-Driven Game Narrative",
      "authors": [
        "Xiangyu Peng",
        "Jessica Quaye",
        "Sudha Rao",
        "Weijia Xu",
        "Portia Botchway",
        "Chris Brockett",
        "Nebojsa Jojic",
        "Gabriel DesGarennes",
        "Ken Lobb",
        "Michael Xu",
        "Jorge Leandro",
        "Claire Jin",
        "Bill Dolan"
      ],
      "date": "2024-04-25",
      "venue": "IEEE Conference on Games 2024",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2404.17027",
      "doi": "10.1109/CoG60054.2024.10645607",
      "primaryUrl": "https://arxiv.org/abs/2404.17027",
      "citations": 34,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Emergent narrative from player interaction with GPT-4 NPCs in a mystery text adventure (28 players).",
      "headlineFinding": "Players discovered new story nodes not in the original narrative; the most \"emergent\" players were those who like exploration games.",
      "axes": [
        "agency",
        "signature"
      ],
      "limits": "One game, small sample.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2601.11529",
      "kind": "paper",
      "group": "agency",
      "title": "SNAP: A Plan-Driven Framework for Controllable Interactive Narrative Generation",
      "authors": [
        "Geonwoo Bang",
        "DongMyung Kim",
        "Hayoung Oh"
      ],
      "date": "2025-11-18",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2601.11529",
      "doi": "10.48550/arXiv.2601.11529",
      "primaryUrl": "https://arxiv.org/abs/2601.11529",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Plan-driven control of interactive narratives to prevent drift (SNAP).",
      "headlineFinding": "Splitting the story into planned cells kept dialogue scenario-consistent under varied user input.",
      "axes": [
        "agency",
        "memory"
      ],
      "limits": "Short paper.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2412.10582",
      "kind": "paper",
      "group": "agency",
      "title": "WHAT-IF: Exploring Branching Narratives by Meta-Prompting Large Language Models",
      "authors": [
        "Runsheng \"Anson\" Huang",
        "Lara J. Martin",
        "Chris Callison-Burch"
      ],
      "date": "2024-12-13",
      "venue": "Published in Wordplay: When Language Meets Games Workshop (EMNLP 2025)",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2412.10582",
      "doi": "10.48550/arXiv.2412.10582",
      "primaryUrl": "https://arxiv.org/abs/2412.10582",
      "citations": 4,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Branching interactive fiction generated from a linear story by meta-prompting (WHAT-IF).",
      "headlineFinding": "Storing the branch tree as a graph kept alternate storylines coherent.",
      "axes": [
        "agency"
      ],
      "limits": "Choice-based IF, not free text.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2311.09213",
      "kind": "paper",
      "group": "agency",
      "title": "GENEVA: GENErating and Visualizing branching narratives using LLMs",
      "authors": [
        "Jorge Leandro",
        "Sudha Rao",
        "Michael Xu",
        "Weijia Xu",
        "Nebosja Jojic",
        "Chris Brockett",
        "Bill Dolan"
      ],
      "date": "2023-11-15",
      "venue": "Accepted at IEEE Conference on Games 2024",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2311.09213",
      "doi": "10.1109/CoG60054.2024.10645625",
      "primaryUrl": "https://arxiv.org/abs/2311.09213",
      "citations": 17,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "LLM tool that generates branching and reconverging narrative graphs for designers (GENEVA).",
      "headlineFinding": "GPT-4 produced rich branching narratives under designer constraints.",
      "axes": [
        "agency",
        "signature"
      ],
      "limits": "Authoring tool, not a runtime GM.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2601.18785",
      "kind": "paper",
      "group": "agency",
      "title": "Design Techniques for LLM-Powered Interactive Storytelling: A Case Study of the Dramamancer System",
      "authors": [
        "Tiffany Wang",
        "Yuqian Sun",
        "Yi Wang",
        "Melissa Roemmele",
        "John Joon Young Chung",
        "Max Kreminski"
      ],
      "date": "2026-01-26",
      "venue": "Wordplay Workshop at EMNLP",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2601.18785",
      "doi": "10.48550/arXiv.2601.18785",
      "primaryUrl": "https://arxiv.org/abs/2601.18785",
      "citations": 1,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Design techniques for turning author-written story schemas into player-driven play (Dramamancer).",
      "headlineFinding": "Frames the author-intent vs player-agency balance as a design problem.",
      "axes": [
        "agency",
        "signature"
      ],
      "limits": "Extended abstract.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2304.03442",
      "kind": "paper",
      "group": "agents",
      "title": "Generative Agents: Interactive Simulacra of Human Behavior",
      "authors": [
        "Joon Sung Park",
        "Joseph C. O'Brien",
        "Carrie J. Cai",
        "Meredith Ringel Morris",
        "Percy Liang",
        "Michael S. Bernstein"
      ],
      "date": "2023-04-07",
      "venue": "ACM Symposium on User Interface Software and Technology",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2304.03442",
      "doi": "10.1145/3586183.3606763",
      "primaryUrl": "https://arxiv.org/abs/2304.03442",
      "citations": 5668,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Believable agents with memory, reflection and planning in a Sims-like town (Generative Agents).",
      "headlineFinding": "The foundational architecture — store experiences, reflect, retrieve to plan — behind most \"AI NPC\" work since.",
      "axes": [
        "npc",
        "memory"
      ],
      "limits": "Social simulation, not RPG play.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2506.03610",
      "kind": "paper",
      "group": "agents",
      "title": "Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games",
      "authors": [
        "Dongmin Park",
        "Minkyu Kim",
        "Beongjun Choi",
        "Junhyuck Kim",
        "Keon Lee",
        "Jonghyun Lee",
        "Inkyu Park",
        "Byeong-Uk Lee",
        "Jaeyoung Hwang",
        "Jaewoo Ahn",
        "Ameya S. Mahabaleshwarkar",
        "Bilal Kartal",
        "Pritam Biswas",
        "Yoshi Suhara",
        "Kangwook Lee",
        "Jaewoong Cho"
      ],
      "date": "2025-06-04",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2506.03610",
      "doi": "10.48550/arXiv.2506.03610",
      "primaryUrl": "https://arxiv.org/abs/2506.03610",
      "citations": 28,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "LLM agents playing 12 popular video games across genres (Orak).",
      "headlineFinding": "Plug-and-play benchmark and fine-tuning data for game-playing agents.",
      "axes": [],
      "limits": "Measures AI as the player, not as the game master.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2503.06047",
      "kind": "paper",
      "group": "agents",
      "title": "DSGBench: A Diverse Strategic Game Benchmark for Evaluating LLM-based Agents in Complex Decision-Making Environments",
      "authors": [
        "Wenjie Tang",
        "Yuan Zhou",
        "Erqiang Xu",
        "Keyan Cheng",
        "Minne Li",
        "Liquan Xiao"
      ],
      "date": "2025-03-08",
      "venue": "IEEE International Conference on Acoustics, Speech, and Signal Processing",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2503.06047",
      "doi": "10.48550/arXiv.2503.06047",
      "primaryUrl": "https://arxiv.org/abs/2503.06047",
      "citations": 21,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Strategic decision-making in six complex games (DSGBench).",
      "headlineFinding": "Fine-grained scoring of agents in long-horizon strategy games.",
      "axes": [],
      "limits": "AI as player.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2504.14128",
      "kind": "paper",
      "group": "agents",
      "title": "TALES: Text Adventure Learning Environment Suite",
      "authors": [
        "Christopher Zhang Cui",
        "Xingdi Yuan",
        "Ziang Xiao",
        "Prithviraj Ammanabrolu",
        "Marc-Alexandre Côté"
      ],
      "date": "2025-04-19",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2504.14128",
      "doi": "10.48550/arXiv.2504.14128",
      "primaryUrl": "https://arxiv.org/abs/2504.14128",
      "citations": 11,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "LLMs playing synthetic and human-written text adventures (TALES).",
      "headlineFinding": "Even top agents score under 15% on text adventures designed for human enjoyment.",
      "axes": [
        "memory"
      ],
      "limits": "AI as player.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2304.02868",
      "kind": "paper",
      "group": "agents",
      "title": "Can Large Language Models Play Text Games Well? Current State-of-the-Art and Open Questions",
      "authors": [
        "Chen Feng Tsai",
        "Xiaochen Zhou",
        "Sierra S. Liu",
        "Jing Li",
        "Mo Yu",
        "Hongyuan Mei"
      ],
      "date": "2023-04-06",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2304.02868",
      "doi": "10.48550/arXiv.2304.02868",
      "primaryUrl": "https://arxiv.org/abs/2304.02868",
      "citations": 39,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "How well ChatGPT plays text games (2023).",
      "headlineFinding": "ChatGPT could not build a world model from play or even the manual.",
      "axes": [
        "memory"
      ],
      "limits": "Early model generation.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2409.12889",
      "kind": "paper",
      "group": "agents",
      "title": "Can VLMs Play Action Role-Playing Games? Take Black Myth Wukong as a Study Case",
      "authors": [
        "Peng Chen",
        "Pi Bu",
        "Jun Song",
        "Yuan Gao",
        "Bo Zheng"
      ],
      "date": "2024-09-19",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2409.12889",
      "doi": "10.48550/arXiv.2409.12889",
      "primaryUrl": "https://arxiv.org/abs/2409.12889",
      "citations": 31,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Vision-language agents playing Black Myth: Wukong.",
      "headlineFinding": "Explores playing an action RPG from screen input only.",
      "axes": [],
      "limits": "Combat play, not narrative.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2504.11442",
      "kind": "paper",
      "group": "agents",
      "title": "TextArena",
      "authors": [
        "Leon Guertler",
        "Bobby Cheng",
        "Simon Yu",
        "Bo Liu",
        "Leshem Choshen",
        "Cheston Tan"
      ],
      "date": "2025-04-15",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2504.11442",
      "doi": "10.48550/arXiv.2504.11442",
      "primaryUrl": "https://arxiv.org/abs/2504.11442",
      "citations": 7,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Competitive text games for agent evaluation (TextArena, 57+ environments).",
      "headlineFinding": "Online leaderboard with TrueSkill for social skills like negotiation and deception.",
      "axes": [],
      "limits": "AI as player.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2509.23979",
      "kind": "paper",
      "group": "agents",
      "title": "ByteSized32Refactored: Towards an Extensible Interactive Text Games Corpus for LLM World Modeling and Evaluation",
      "authors": [
        "Haonan Wang",
        "Junfeng Sun",
        "Xingdi Yuan",
        "Ruoyao Wang",
        "Ziang Xiao"
      ],
      "date": "2025-09-28",
      "venue": "Accepted to the 5th Wordplay: When Language Meets Games Workshop, EMNLP 2025",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2509.23979",
      "doi": "10.48550/arXiv.2509.23979",
      "primaryUrl": "https://arxiv.org/abs/2509.23979",
      "citations": 0,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Corpus of 32 text games for LLM world modelling (ByteSized32Refactored).",
      "headlineFinding": "Refactored, extensible text-game corpus for generation and evaluation.",
      "axes": [],
      "limits": "Resource paper.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2404.18231",
      "kind": "paper",
      "group": "survey",
      "title": "From Persona to Personalization: A Survey on Role-Playing Language Agents",
      "authors": [
        "Jiangjie Chen",
        "Xintao Wang",
        "Rui Xu",
        "Siyu Yuan",
        "Yikai Zhang",
        "Wei Shi",
        "Jian Xie",
        "Shuang Li",
        "Ruihan Yang",
        "Tinghui Zhu",
        "Aili Chen",
        "Nianqi Li",
        "Lida Chen",
        "Caiyu Hu",
        "Siye Wu",
        "Scott Ren",
        "Ziquan Fu",
        "Yanghua Xiao"
      ],
      "date": "2024-04-28",
      "venue": "Accepted to TMLR 2024",
      "venueSource": "arXiv comment/journal-ref (author-stated)",
      "arxivId": "2404.18231",
      "doi": "10.48550/arXiv.2404.18231",
      "primaryUrl": "https://arxiv.org/abs/2404.18231",
      "citations": 289,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Survey of role-playing language agents: demographic, character and individualised personas.",
      "headlineFinding": "The most-cited field map of role-playing agents.",
      "axes": [
        "npc",
        "memory"
      ],
      "limits": "General role-play, games are one application.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2601.10122",
      "kind": "paper",
      "group": "survey",
      "title": "Role-Playing Agents Driven by Large Language Models: Current Status, Challenges, and Future Trends",
      "authors": [
        "Ye Wang",
        "Jiaxing Chen",
        "Hongjiang Xiao"
      ],
      "date": "2026-01-15",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2601.10122",
      "doi": "10.48550/arXiv.2601.10122",
      "primaryUrl": "https://arxiv.org/abs/2601.10122",
      "citations": 2,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "2026 review of role-playing agents: personality modelling, memory, evaluation.",
      "headlineFinding": "Traces the field from templates to cognitive simulation.",
      "axes": [
        "npc",
        "memory"
      ],
      "limits": "Review.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2407.11484",
      "kind": "paper",
      "group": "survey",
      "title": "The Oscars of AI Theater: A Survey on Role-Playing with Language Models",
      "authors": [
        "Nuo Chen",
        "Yan Wang",
        "Yang Deng",
        "Jia Li"
      ],
      "date": "2024-07-16",
      "venue": "arXiv preprint",
      "venueSource": "arXiv",
      "arxivId": "2407.11484",
      "doi": "10.48550/arXiv.2407.11484",
      "primaryUrl": "https://arxiv.org/abs/2407.11484",
      "citations": 64,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Survey of role-playing with language models (\"The Oscars of AI Theater\").",
      "headlineFinding": "Taxonomy of data, models, architecture and evaluation for role-play.",
      "axes": [
        "npc"
      ],
      "limits": "Review.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "arxiv-2502.13012",
      "kind": "paper",
      "group": "survey",
      "title": "Towards a Design Guideline for RPA Evaluation: A Survey of Large Language Model-Based Role-Playing Agents",
      "authors": [
        "Chaoran Chen",
        "Bingsheng Yao",
        "Ruishi Zou",
        "Wenyue Hua",
        "Weimin Lyu",
        "Yanfang Ye",
        "Toby Jia-Jun Li",
        "Dakuo Wang"
      ],
      "date": "2025-02-18",
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "venueSource": "Semantic Scholar (unconfirmed by authors on arXiv)",
      "arxivId": "2502.13012",
      "doi": "10.48550/arXiv.2502.13012",
      "primaryUrl": "https://arxiv.org/abs/2502.13012",
      "citations": 31,
      "citationsCheckedOn": "2026-09-25",
      "whatItMeasures": "Evaluation-design guideline from a review of 1,676 role-playing agent papers.",
      "headlineFinding": "Identifies six agent attributes, seven task attributes and seven evaluation metrics used across the literature.",
      "axes": [
        "npc"
      ],
      "limits": "Guideline for researchers.",
      "verifiedOn": "2026-09-25",
      "verifiedFrom": "arXiv API (title, authors, date, abstract) + Semantic Scholar (citation count, venue)"
    },
    {
      "id": "nvidia-ace",
      "kind": "tech",
      "group": "tech",
      "title": "NVIDIA ACE",
      "vendor": "NVIDIA",
      "primaryUrl": "https://developer.nvidia.com/ace",
      "status": "shipping (SDK)",
      "whatItIs": "NVIDIA describes ACE as \"a suite of digital human technologies that power agentic workflows for autonomous game characters and digital assistants.\"",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "vendor developer page",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "inworld",
      "kind": "tech",
      "group": "tech",
      "title": "Inworld AI",
      "vendor": "Inworld",
      "primaryUrl": "https://inworld.ai/",
      "status": "pivoted",
      "whatItIs": "Once known for game-character AI; its homepage now sells \"Realtime TTS and STT models, LLM serving, and the inference behind both\" for consumer-facing applications.",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "vendor homepage",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "convai",
      "kind": "tech",
      "group": "tech",
      "title": "Convai",
      "vendor": "Convai",
      "primaryUrl": "https://convai.com/",
      "status": "shipping",
      "whatItIs": "Positions itself as \"Conversational AI for Virtual Worlds.\"",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "vendor homepage title",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "ubisoft-neo-npc",
      "kind": "tech",
      "group": "tech",
      "title": "Ubisoft NEO NPC (R&D)",
      "vendor": "Ubisoft",
      "primaryUrl": "https://news.ubisoft.com/en-us/article/5qXdxhshJBXoanFZApdG3L",
      "status": "research prototype",
      "whatItIs": "A small R&D team at Ubisoft Paris experimenting with generative AI toward real conversations with NPCs, per Ubisoft's own news post.",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "Ubisoft News",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "mantella",
      "kind": "mod",
      "group": "tech",
      "title": "Mantella",
      "vendor": "art-from-the-machine (open source)",
      "primaryUrl": "https://github.com/art-from-the-machine/Mantella",
      "status": "active (repo pushed 2026-07-16)",
      "whatItIs": "A Skyrim and Fallout 4 mod \"which allows you to naturally speak to NPCs using a Speech-to-Text → LLMs → Text-to-Speech pipeline.\"",
      "axes": [
        "npc",
        "memory"
      ],
      "verifiedFrom": "GitHub repository",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-ai-roguelite",
      "kind": "game",
      "group": "tech",
      "title": "AI Roguelite",
      "vendor": "Steam app 1889620",
      "primaryUrl": "https://store.steampowered.com/app/1889620/",
      "status": "released 2023-10-25",
      "whatItIs": "Developer disclosure on Steam: it \"heavily uses AI to live-generate in-game content such as text, images, and sound effects\" and \"to make a variety of game mechanics decisions in real time.\"",
      "axes": [
        "depth",
        "signature"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-inzoi",
      "kind": "game",
      "group": "tech",
      "title": "inZOI",
      "vendor": "Steam app 2456740 (Krafton)",
      "primaryUrl": "https://store.steampowered.com/app/2456740/",
      "status": "early access since 2025-03-27",
      "whatItIs": "Steam disclosure: player text can influence \"character actions and thoughts … using SLM technology\", plus AI-generated textures, 3D objects and motions.",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-where-winds-meet",
      "kind": "game",
      "group": "tech",
      "title": "Where Winds Meet",
      "vendor": "Steam app 3564740",
      "primaryUrl": "https://store.steampowered.com/app/3564740/",
      "status": "released 2025-11-14",
      "whatItIs": "Steam disclosure: \"AI-driven NPC text and voice chat that responds to player input in real time.\"",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-suck-up",
      "kind": "game",
      "group": "tech",
      "title": "Suck Up!",
      "vendor": "Steam app 2726370 (Proxima)",
      "primaryUrl": "https://store.steampowered.com/app/2726370/",
      "status": "released 2025-10-01",
      "whatItIs": "Steam disclosure: players talk to AI characters by voice and \"the AI responds in real time based on tone and strategy.\"",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-wanderfolk",
      "kind": "game",
      "group": "tech",
      "title": "Wanderfolk",
      "vendor": "Steam app 4599270 (Kinetic Sky)",
      "primaryUrl": "https://store.steampowered.com/app/4599270/",
      "status": "coming 2026",
      "whatItIs": "Steam disclosure: villager conversations \"generated in real time by a large language model\", and NPCs \"remember what you've said to them across the playthrough.\"",
      "axes": [
        "npc",
        "memory"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    },
    {
      "id": "steam-vaudeville",
      "kind": "game",
      "group": "tech",
      "title": "Vaudeville",
      "vendor": "Steam app 2240920 (Bumblebee Studios)",
      "primaryUrl": "https://store.steampowered.com/app/2240920/",
      "status": "released 2025-11-28",
      "whatItIs": "Steam disclosure: \"dialogues generated with the help of a conversational AI\"; players talk to townsfolk by typing or voice.",
      "axes": [
        "npc"
      ],
      "verifiedFrom": "Steam store page AI disclosure",
      "verifiedOn": "2026-09-25"
    }
  ]
}
