{"as_of":"2026-08-15T13:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f78a1431aea9bb613bde63de206cb54cd212c52f7c99e447c8d823c941336434","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:24:03.892023Z","state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.08160/citation-record","integrity":"/paper/2608.08160/integrity","json":"/paper/2608.08160/citation-record.json","paper":"/paper/2608.08160"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2001.09977","last_updated":"2020-02-27T07:36:47Z","snapshot_observed_at":"2026-08-04T13:03:06.820772Z","submitted_at":"2020-01-27T18:53:15Z","title":"Towards a Human-like Open-Domain Chatbot","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.09977","snapshot_observed_at":"2026-08-12T00:24:03.785496Z","title":"R., Hall, J., Fiedel, N., Thoppilan, R., Yang, Z., Kulshreshtha, A., Nemade, G., Lu, Y ., et al","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.785496Z"},"links":{"cited_paper":"/paper/2001.09977","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:e565d771a94148483ae1be049b9a9b72a8e2bc7b59b90488733aee9e1f99dca3","observation_id":"6c978dad-8c8e-4668-9f37-f53f8112b1b7","resolution":{"observed_at":"2026-08-12T00:24:03.785496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.352078Z","title":"Prometheus: Inducing fine-grained evaluation capability in language models","venue":null,"work_id":"c979df79-bf15-43bf-8675-b552b1870281","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.802957Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:79790670a7a9aede1ab8d5bf587d80755ca1c96c5132c9048f492843e8b6c107","observation_id":"c6a18712-fc71-414b-ad6f-898742112cba","resolution":{"observed_at":"2026-08-12T00:24:04.357075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.337159Z","title":"Checkeval: A reliable llm-as-a-judge framework for evaluating text generation using checklists","venue":null,"work_id":"0569b6c5-c895-42b4-80ed-c20e1243725e","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.808280Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:d45d69990f40eae5576ed5f04984a906852a4b00c498f1b10374eeccb459fafd","observation_id":"36f5f8fc-770d-46e7-afd7-39ad98dcd032","resolution":{"observed_at":"2026-08-12T00:24:04.341954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-08-12T08:24:59.243273Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-12T00:24:03.813314Z","title":"Deepseek-v3","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.813314Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:863cdfa969014ad83c55e1ecce8a9150ddc3e9b52478ee2fda1319d1bec8aea8","observation_id":"68b75013-c94b-4c02-ac03-8690e3070cb1","resolution":{"observed_at":"2026-08-12T00:24:03.813314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.303740Z","title":"Player-driven emergence in llm-driven game narrative","venue":null,"work_id":"060c0b7d-5f74-4e2a-bd2d-e449950965b9","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.823848Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:0ac52a2455666e409102780e14e94319e925d8494b1aabc861fc266359eddbdb","observation_id":"1607fd6b-28bb-41ab-a7b5-e57cda87559e","resolution":{"observed_at":"2026-08-12T00:24:04.308832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.08239","last_updated":"2022-02-10T16:30:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-20T15:44:37Z","title":"LaMDA: Language Models for Dialog Applications","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.08239","snapshot_observed_at":"2026-08-12T00:24:03.860945Z","title":"Lamda: Language models for dialog appli- cations.arXiv preprint arXiv:2201.08239,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.860945Z"},"links":{"cited_paper":"/paper/2201.08239","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:c87e21aeaf4611de74ab3c21463296485b1d705440657063e812f2d12fc35b21","observation_id":"35975542-e343-4935-81f1-f10bde27460e","resolution":{"observed_at":"2026-08-12T00:24:03.860945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.194268Z","title":"Wang, L., Lian, J., Huang, Y ., Dai, Y ., Li, H., Chen, X., Xie, X., and Wen, J.-R","venue":null,"work_id":"2a3b1c03-4d7d-4770-8c38-f118715231f8","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.872129Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:860aea499572fffa1b9c8ff089a3e7d62a789ff9440b9f5e83675d8a547d0239","observation_id":"b551a7ed-98ce-4ea7-8349-3f2efffd131a","resolution":{"observed_at":"2026-08-12T00:24:04.199992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01489","last_updated":"2024-10-29T17:29:27Z","snapshot_observed_at":"2026-08-10T19:28:41.964638Z","submitted_at":"2024-07-01T17:24:45Z","title":"Agentless: Demystifying LLM-based Software Engineering Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01489","snapshot_observed_at":"2026-08-12T00:24:03.876817Z","title":"ISBN 979-8-89176-251-0","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.876817Z"},"links":{"cited_paper":"/paper/2407.01489","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:b259046de04a9db0fdafa615e33f0cb0090c9bcffff81188925add807224c3c3","observation_id":"0440e59e-1a60-4e6d-ab2d-0535c010162d","resolution":{"observed_at":"2026-08-12T00:24:03.876817Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-12T00:24:03.881671Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.881671Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:be85628e5ecd96710bea18e39d5093646743dd6ab1d4034f5ba099c493b99b3e","observation_id":"4b20e77d-dc52-46a8-a0c7-f6ffda46cd4c","resolution":{"observed_at":"2026-08-12T00:24:03.881671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:03.887012Z","title":"Score: Story coherence and retrieval enhancement for ai narratives","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.887012Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:9f99624d37de6c5d0c8be67689e55bbdfdb23649c809e48c94bbb417a5d39816","observation_id":"bec70c5d-a753-4fa3-96f0-94225235b708","resolution":{"observed_at":"2026-08-12T00:24:03.887012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.176489Z","title":"Interesting","venue":null,"work_id":"ab80e554-741d-4659-a9a1-622973d5486d","year":2003},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.892023Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:536bf55c6d51e9cf473039527e5524e4fe1dae37b2918653b86cfd910eb6463d","observation_id":"134ab024-c815-485d-b717-19c6318be0a9","resolution":{"observed_at":"2026-08-12T00:24:04.182847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.320004Z","title":"Self- contradictory hallucinations of large language models: Evaluation, detection and mitigation","venue":null,"work_id":"86e196bb-e88e-49cc-9db7-7aed40134486","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2003,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.819040Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:7fc1556599982f8c95edc89df895a6db07ad9fbfb227037c12e6ffd32d28e0bb","observation_id":"ba62213a-19f4-414a-bb3f-09d14ba10607","resolution":{"observed_at":"2026-08-12T00:24:04.325603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.251564Z","title":"What makes a good conversation? how controllable attributes affect hu- man judgments","venue":null,"work_id":"d1fce346-67e0-4d01-9285-3ae864b5485f","year":2019},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.839707Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:35060706d51e3e150c194693146f54cac9ed71d28772869ca2a8ed4cee022f54","observation_id":"d4cb5e96-c8f0-437a-9a00-e67905585a6a","resolution":{"observed_at":"2026-08-12T00:24:04.257169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.270361Z","title":"Plot- machines: Outline-conditioned generation with dynamic plot state tracking","venue":null,"work_id":"fcbe916c-e471-4814-8ffc-1e7d70951972","year":2020},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.833990Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:b34747aaf1c9ddcebc35300befe3c5a5d4e400593606e6e416dfe7263c33032c","observation_id":"54b0f562-912b-44ee-b09f-f9bd56f901b5","resolution":{"observed_at":"2026-08-12T00:24:04.275910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-08-12T00:25:39.214406Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-12T00:24:03.850635Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.850635Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:b97cdd5ea85fd4b3d1e54ce94bad4f4aad7df3e84a399907f12d42e0b5a97eb6","observation_id":"c7051bb3-133e-4697-b26d-bb735d28ebf8","resolution":{"observed_at":"2026-08-12T00:24:03.850635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09099","last_updated":"2025-01-15T19:32:32Z","snapshot_observed_at":"2026-08-14T11:14:03.649384Z","submitted_at":"2025-01-15T19:32:32Z","title":"Drama Llama: An LLM-Powered Storylets Framework for Authorable Responsiveness in Interactive Narrative","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09099","snapshot_observed_at":"2026-08-12T00:24:03.844992Z","title":"J., Chung, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.844992Z"},"links":{"cited_paper":"/paper/2501.09099","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:5c8bf7d27eb257803822de777a1bc60846561ef283df2c369c9aca644fcf2a1b","observation_id":"8425bebe-4687-4561-8b98-44732b6b51ad","resolution":{"observed_at":"2026-08-12T00:24:03.844992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.216447Z","title":"Are large language models capable of generating human-level narratives? InPro- ceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp","venue":null,"work_id":"4da972f6-bc83-4240-a637-f5c37f8562c5","year":2024},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.867337Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:ce03108d7339d5e02fc94ee393a328b96d20c5587d4907b63cc0afad2259d4ff","observation_id":"b50cad13-f1d4-4bbc-ae47-7c4e9bea53ef","resolution":{"observed_at":"2026-08-12T00:24:04.222720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-12T00:24:03.791845Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities.arXiv preprint arXiv:2507.06261,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.791845Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:3c87545b3a20a1d9c76a6f5aaa4126c1cf2c3e0a17c3785a27937aa33eabce93","observation_id":"a74d2641-a08f-4694-a266-9d1200787481","resolution":{"observed_at":"2026-08-12T00:24:03.791845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.287745Z","title":"Red teaming language models with language models","venue":null,"work_id":"442ac892-58d8-445d-b1ef-deb985b264a0","year":2022},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.828593Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:f4bfa54f69680cd7d9a01d592b9b959fce477ed2144253a1a7a60d01716607d8","observation_id":"8743f3e3-b051-4735-9fe1-29456f6b55b6","resolution":{"observed_at":"2026-08-12T00:24:04.292768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T05:02:49.183338Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-12T00:24:03.797463Z","title":"P., Perelman, A., Ramesh, A., Clark, A., Ostrow, A., Welihinda, A., Hayes, A., Radford, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.797463Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:06b0df703ca88c14a282332d7ad8f7fb5933b0d74ff75641bd5da017859ee038","observation_id":"5464d4e5-f611-466d-8cb5-7484c48e3df1","resolution":{"observed_at":"2026-08-12T00:24:03.797463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:24:04.234900Z","title":"T., Liu, H., Liu, T., Wang, C., Liu, T., Zhang, Y ., Shipman, F., et al","venue":null,"work_id":"12a1cdb5-68de-4f85-ba58-a33f4db1e6bb","year":2025},"citing_paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-12T00:24:03.856261Z"},"links":{"citing_paper":"/paper/2608.08160"},"observation_digest":"sha256:e6614fb1a181f1ba873dd1c72ebac34ca57477ee61d92265d6baace8f9c6143e","observation_id":"b3ec1cb2-5686-4179-a1b6-ab08ad2558fa","resolution":{"observed_at":"2026-08-12T00:24:04.239975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.08160","last_updated":"2026-08-08T14:38:18Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T23:24:17.956219Z","submitted_at":"2026-08-08T14:38:18Z","title":"Can LLM Agents Stick to the Script? A Benchmark for Long-Horizon Consistency in Interactive Narratives"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":11},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 0 inbound Pith citation observations for arXiv:2608.08160."}