{"as_of":"2026-08-13T02:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:70a79f2571b3ff5a1df6f579fa39a48a6b6f7a1e1619e47d79102e9fcc38e278","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:24:03.892023Z","state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.08160/citation-record","integrity":"/paper/2608.08160/integrity","json":"/paper/2608.08160/citation-record.json","paper":"/paper/2608.08160"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2001.09977","last_updated":"2020-02-27T07:36:47Z","snapshot_observed_at":"2026-08-04T13:03:06.820772Z","submitted_at":"2020-01-27T18:53:15Z","title":"Towards a Human-like Open-Domain Chatbot","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.09977","snapshot_observed_at":"2026-08-12T00:24:03.785496Z","title":"R., Hall, J., Fiedel, N., Thoppilan, R., Yang, Z., Kulshreshtha, A., Nemade, G., Lu, Y ., et al","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.785496Z"},"links":{"cited_paper":"/paper/2001.09977","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:7efff85983cd5eb34829417812071cbf845cd7eef503c3031f414ab760ee2f07","observation_id":"6c978dad-8c8e-4668-9f37-f53f8112b1b7","resolution":{"observed_at":"2026-08-12T00:24:03.785496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.352078Z","title":"Prometheus: Inducing fine-grained evaluation capability in language models","venue":null,"work_id":"c979df79-bf15-43bf-8675-b552b1870281","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.802957Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:6ca45015910b4b718d20f406914cb490a74e82d8b21145ee2abb7ee1787693ba","observation_id":"c6a18712-fc71-414b-ad6f-898742112cba","resolution":{"observed_at":"2026-08-12T00:24:04.357075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.337159Z","title":"Checkeval: A reliable llm-as-a-judge framework for evaluating text generation using checklists","venue":null,"work_id":"0569b6c5-c895-42b4-80ed-c20e1243725e","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.808280Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:2f682cdb27d6f7b2f482b610a0eabbcc4a26c94ffff4b67cfba050148efef4e7","observation_id":"36f5f8fc-770d-46e7-afd7-39ad98dcd032","resolution":{"observed_at":"2026-08-12T00:24:04.341954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-08-12T08:24:59.243273Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-12T00:24:03.813314Z","title":"Deepseek-v3","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.813314Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:5d8da8b07673f778c4745215843728e8357259b7fd4c944662dd9ed020ae803a","observation_id":"68b75013-c94b-4c02-ac03-8690e3070cb1","resolution":{"observed_at":"2026-08-12T00:24:03.813314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.303740Z","title":"Player-driven emergence in llm-driven game narrative","venue":null,"work_id":"060c0b7d-5f74-4e2a-bd2d-e449950965b9","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.823848Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:38c0934b6b9dfbd00a3e4ab1f8c64d3885f114b896dd26d1bc6fe4dca531b788","observation_id":"1607fd6b-28bb-41ab-a7b5-e57cda87559e","resolution":{"observed_at":"2026-08-12T00:24:04.308832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.08239","last_updated":"2022-02-10T16:30:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-20T15:44:37Z","title":"LaMDA: Language Models for Dialog Applications","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.08239","snapshot_observed_at":"2026-08-12T00:24:03.860945Z","title":"Lamda: Language models for dialog appli- cations.arXiv preprint arXiv:2201.08239,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.860945Z"},"links":{"cited_paper":"/paper/2201.08239","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:5a566604e4304bedd50b9363d0a486b1f518697eaf0b94209f39ae852d099a4f","observation_id":"35975542-e343-4935-81f1-f10bde27460e","resolution":{"observed_at":"2026-08-12T00:24:03.860945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.194268Z","title":"Wang, L., Lian, J., Huang, Y ., Dai, Y ., Li, H., Chen, X., Xie, X., and Wen, J.-R","venue":null,"work_id":"2a3b1c03-4d7d-4770-8c38-f118715231f8","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.872129Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:8a287b9095e29dd74fcd4e8f80932c52a5975e4d72b88d15327fce8096e13068","observation_id":"b551a7ed-98ce-4ea7-8349-3f2efffd131a","resolution":{"observed_at":"2026-08-12T00:24:04.199992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01489","last_updated":"2024-10-29T17:29:27Z","snapshot_observed_at":"2026-08-10T19:28:41.964638Z","submitted_at":"2024-07-01T17:24:45Z","title":"Agentless: Demystifying LLM-based Software Engineering Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01489","snapshot_observed_at":"2026-08-12T00:24:03.876817Z","title":"ISBN 979-8-89176-251-0","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.876817Z"},"links":{"cited_paper":"/paper/2407.01489","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:b0b3e1719123889325b40b4aec277f67823319296b770ff762c2e55616cdc007","observation_id":"0440e59e-1a60-4e6d-ab2d-0535c010162d","resolution":{"observed_at":"2026-08-12T00:24:03.876817Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-12T00:24:03.881671Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.881671Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:af6f971918e398117d906a4a5bc8ccaaae43f8bc27979cc06c4f98bd58936f3e","observation_id":"4b20e77d-dc52-46a8-a0c7-f6ffda46cd4c","resolution":{"observed_at":"2026-08-12T00:24:03.881671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:03.887012Z","title":"Score: Story coherence and retrieval enhancement for ai narratives","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.887012Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:f6b04436cbcf32308b4e6b5d35b0c4f28e96590d3574e20251d1bae0f27b234f","observation_id":"bec70c5d-a753-4fa3-96f0-94225235b708","resolution":{"observed_at":"2026-08-12T00:24:03.887012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.176489Z","title":"Interesting","venue":null,"work_id":"ab80e554-741d-4659-a9a1-622973d5486d","year":2003},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.892023Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:ccfdeabea49aa89f61be2ee33cb8619acc93eb2d0a619b025b009cb4bd2c0c8f","observation_id":"134ab024-c815-485d-b717-19c6318be0a9","resolution":{"observed_at":"2026-08-12T00:24:04.182847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.320004Z","title":"Self- contradictory hallucinations of large language models: Evaluation, detection and mitigation","venue":null,"work_id":"86e196bb-e88e-49cc-9db7-7aed40134486","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2003,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.819040Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:fce38a5b971f54220fd9425d9a6a1c43de7f5b3a352e641d661b3d10c5ae9f6d","observation_id":"ba62213a-19f4-414a-bb3f-09d14ba10607","resolution":{"observed_at":"2026-08-12T00:24:04.325603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.251564Z","title":"What makes a good conversation? how controllable attributes affect hu- man judgments","venue":null,"work_id":"d1fce346-67e0-4d01-9285-3ae864b5485f","year":2019},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.839707Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:3184c0282948bf91aabf3fc04b7044266eda529f84bd3050fd077491048203e8","observation_id":"d4cb5e96-c8f0-437a-9a00-e67905585a6a","resolution":{"observed_at":"2026-08-12T00:24:04.257169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.270361Z","title":"Plot- machines: Outline-conditioned generation with dynamic plot state tracking","venue":null,"work_id":"fcbe916c-e471-4814-8ffc-1e7d70951972","year":2020},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.833990Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:8fc9299f180d8a8044beb60a3e00c57dbd670e3771b39c530ed8c5430550ff11","observation_id":"54b0f562-912b-44ee-b09f-f9bd56f901b5","resolution":{"observed_at":"2026-08-12T00:24:04.275910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-08-12T00:25:39.214406Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-12T00:24:03.850635Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.850635Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:6ab0d39060faaa885103154e234230cbfeadbbd2a5f49ff8ea977e821a835bc2","observation_id":"c7051bb3-133e-4697-b26d-bb735d28ebf8","resolution":{"observed_at":"2026-08-12T00:24:03.850635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09099","last_updated":"2025-01-15T19:32:32Z","snapshot_observed_at":"2026-08-13T00:43:21.294748Z","submitted_at":"2025-01-15T19:32:32Z","title":"Drama Llama: An LLM-Powered Storylets Framework for Authorable Responsiveness in Interactive Narrative","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09099","snapshot_observed_at":"2026-08-12T00:24:03.844992Z","title":"J., Chung, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.844992Z"},"links":{"cited_paper":"/paper/2501.09099","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:1e82d41cc64ac485e52709e548e07e1eb5b863815df5eb179d6cdfedffc78f87","observation_id":"8425bebe-4687-4561-8b98-44732b6b51ad","resolution":{"observed_at":"2026-08-12T00:24:03.844992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.216447Z","title":"Are large language models capable of generating human-level narratives? InPro- ceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp","venue":null,"work_id":"4da972f6-bc83-4240-a637-f5c37f8562c5","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.867337Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:f3aa04fd645a18960e827c1c7d21e9c2cfb4a5cbeb1f7581166b0dd896913d51","observation_id":"b50cad13-f1d4-4bbc-ae47-7c4e9bea53ef","resolution":{"observed_at":"2026-08-12T00:24:04.222720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-12T00:24:03.791845Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities.arXiv preprint arXiv:2507.06261,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.791845Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:17464949d3b8d54f1a0aeada16780f04c8337976180d226b0d599851ccf5ac76","observation_id":"a74d2641-a08f-4694-a266-9d1200787481","resolution":{"observed_at":"2026-08-12T00:24:03.791845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.287745Z","title":"Red teaming language models with language models","venue":null,"work_id":"442ac892-58d8-445d-b1ef-deb985b264a0","year":2022},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.828593Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:29042da39e708abc764fafeeec571581e738e51e69da32636194fd4a55ab47f1","observation_id":"8743f3e3-b051-4735-9fe1-29456f6b55b6","resolution":{"observed_at":"2026-08-12T00:24:04.292768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-12T00:24:03.797463Z","title":"P., Perelman, A., Ramesh, A., Clark, A., Ostrow, A., Welihinda, A., Hayes, A., Radford, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.797463Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:0890596f78b2f926f6f4ba072fe1344a822d97cbed17dac384bae9c6ba6d6638","observation_id":"5464d4e5-f611-466d-8cb5-7484c48e3df1","resolution":{"observed_at":"2026-08-12T00:24:03.797463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.234900Z","title":"T., Liu, H., Liu, T., Wang, C., Liu, T., Zhang, Y ., Shipman, F., et al","venue":null,"work_id":"12a1cdb5-68de-4f85-ba58-a33f4db1e6bb","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.856261Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:6ecdb3f7ad2ce37171a5d513b577f52cffdcf4f7240101af311b7a4b2b296449","observation_id":"b3ec1cb2-5686-4179-a1b6-ab08ad2558fa","resolution":{"observed_at":"2026-08-12T00:24:04.239975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T02:11:36.187272Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":11},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 0 inbound Pith citation observations for arXiv:2608.08160."}