{"as_of":"2026-08-08T22:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b1bd628e6c54586f1a4dd129f5b32aacf83594d4852015f40040b69af5b5a89c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:33:12.420401Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:48:56.198096Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2505.06120","last_updated":"2025-05-09T15:21:44Z","snapshot_observed_at":"2026-08-08T19:35:51.229758Z","submitted_at":"2025-05-09T15:21:44Z","title":"LLMs Get Lost In Multi-Turn Conversation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-14T00:57:10.262350Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.06120"},"observation_digest":"sha256:2c4cb6a248ee54d2241bd40ef74d345680b83344bd1eef1c05fe1c8e55b5280c","observation_id":"140c103e-9697-47e7-b78b-0a93bb3b179d","resolution":{"observed_at":"2026-05-14T01:11:09.335166Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-07T15:33:12.420401Z","title":"Multichallenge: A realistic multi-turn conversation evaluation benchmark challenging to frontier llms.arXiv preprint arXiv:2501.17399, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14810","last_updated":"2025-05-25T14:52:51Z","snapshot_observed_at":"2026-08-07T15:27:03.992813Z","submitted_at":"2025-05-20T18:18:01Z","title":"Scaling Reasoning, Losing Control: Evaluating Instruction Following in Large Reasoning Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:33:12.420401Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.14810"},"observation_digest":"sha256:11546a24f4117d39335bbcacd8ade4a2a18d8b94bd50750b29aa8695767f7c9e","observation_id":"6e46e7b3-0481-483d-8932-7b2ab6cebf05","resolution":{"observed_at":"2026-08-07T15:33:12.420401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-07T15:27:01.110876Z","title":"arXiv preprint arXiv:2501.17399 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-07T21:22:21.405277Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.110876Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cc457ef359346577ccccbbfd5b8c09633858c4cb5cf907b747de980a0eb97fc5","observation_id":"8158aad3-a061-4172-a5e1-392d3e4fa3a4","resolution":{"observed_at":"2026-08-07T15:27:01.110876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2505.20275","last_updated":"2025-05-26T17:53:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-26T17:53:33Z","title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-12T18:17:45.123690Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.20275"},"observation_digest":"sha256:2019f0ac30d5fa23db020df72b2600ef4c7936f960c5b64a4da987c690304303","observation_id":"583020f1-6aaa-4059-a310-44715a23cdb8","resolution":{"observed_at":"2026-05-12T18:17:45.350173Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2506.13585","last_updated":"2025-06-16T15:08:02Z","snapshot_observed_at":"2026-08-07T04:50:43.413490Z","submitted_at":"2025-06-16T15:08:02Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-12T09:28:16.189617Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2506.13585"},"observation_digest":"sha256:64a0a125a095414343f2b602cbc4c0a477483cdf428b1344296defd525ef7d25","observation_id":"0e6f16b8-a732-4ca0-90e5-adf7914b8582","resolution":{"observed_at":"2026-05-12T09:28:16.518006Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-06T23:25:51.856959Z","title":"Multichallenge: A realistic multi-turn conversation evaluation benchmark challenging to frontier llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.18213","last_updated":"2025-06-23T00:19:27Z","snapshot_observed_at":"2026-08-07T10:41:05.932475Z","submitted_at":"2025-06-23T00:19:27Z","title":"A Conceptual Framework for AI Capability Evaluations","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-06T23:25:51.856959Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2506.18213"},"observation_digest":"sha256:45e4c4a6bc006b417f6fed83d55b50debc0710c1f42fc2f68b4ca2a6b8ad4079","observation_id":"c76d5d24-86a7-48e5-a1ca-4b014cc39af5","resolution":{"observed_at":"2026-08-06T23:25:51.856959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:56.678339Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2507.17746"},"observation_digest":"sha256:b2e083e2158acaa0499274c9419184ef887a0600d856239dce1caf9703454f97","observation_id":"b2aa6ed7-d66e-413e-b8f1-d92c97ad80f3","resolution":{"observed_at":"2026-05-13T06:07:56.845701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-06T10:32:33.669131Z","title":"Susan Smetale","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.23701","last_updated":"2025-08-13T17:45:14Z","snapshot_observed_at":"2026-08-08T20:33:26.749462Z","submitted_at":"2025-07-31T16:22:55Z","title":"TextQuests: How Good are LLMs at Text-Based Video Games?","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T10:32:33.669131Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2507.23701"},"observation_digest":"sha256:7e6403dbcd570520ee0feabf82e54b370c3f132cc19de9d97111609db7bbe2a2","observation_id":"0a513bc0-9837-42b3-8f72-4faf142f9ccd","resolution":{"observed_at":"2026-08-06T10:32:33.669131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-05T14:21:48.990246Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02594","last_updated":"2026-07-27T13:01:03Z","snapshot_observed_at":"2026-08-05T14:21:46.209025Z","submitted_at":"2025-08-29T09:51:41Z","title":"OpenAIs HealthBench in Action: Evaluating an LLM-Based Medical Assistant on Realistic Clinical Queries","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T14:21:48.990246Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2509.02594"},"observation_digest":"sha256:52d547eff0ed4c117e558a2538dea4446b15368cec1dc4af9b358cc0db3fcfee","observation_id":"57df649d-5060-47bc-9a32-08469a0b455a","resolution":{"observed_at":"2026-08-05T14:21:48.990246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-04T23:12:10.089168Z","title":"Multichallenge: A realistic multi-turn conversation evaluation benchmark challenging to frontier llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06770","last_updated":"2025-09-13T00:41:46Z","snapshot_observed_at":"2026-08-08T06:08:27.835909Z","submitted_at":"2025-09-08T14:54:31Z","title":"Another Turn, Better Output? A Turn-Wise Analysis of Iterative LLM Prompting","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T23:12:10.089168Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2509.06770"},"observation_digest":"sha256:8a194330ca93979720b040908e914b6fe25b699602bf30ae0c61f23cd34a9b0a","observation_id":"4cfd9f18-3805-4e8e-867d-263f705e7254","resolution":{"observed_at":"2026-08-04T23:12:10.089168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-04T16:07:39.919893Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":152,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:39.919893Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:69ccf3a74de6da8785b3c6d57401c5289cc3bd89441d8d41327f2778488b0fe6","observation_id":"5c091189-e56c-49de-bc03-2054b4e04159","resolution":{"observed_at":"2026-08-04T16:07:39.919893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-04T09:40:46.562258Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.14207","last_updated":"2026-06-29T03:28:16Z","snapshot_observed_at":"2026-08-07T22:39:10.403032Z","submitted_at":"2025-10-16T01:27:44Z","title":"Echoes of Human Malice in Agents: Benchmarking LLMs for Multi-Turn Online Harassment Attacks","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T09:40:46.562258Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2510.14207"},"observation_digest":"sha256:3634a0c75de8fb538b15fe11b96ca8dbef1b33969616e6c7c1df4186f357b2ef","observation_id":"dc68545d-532a-47d9-900e-a2d6861c90f4","resolution":{"observed_at":"2026-08-04T09:40:46.562258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2511.20857","last_updated":"2026-05-18T16:18:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T21:08:07Z","title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","version":1},"reference_index":131,"source":"arxiv_source","source_observed_at":"2026-05-14T23:13:15.016486Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2511.20857"},"observation_digest":"sha256:bbb32b433b904a896f3d1cbde07f6f3905a5b78e8fc6ed2805cb68178c41ec7c","observation_id":"99759f4e-bdf1-4db9-bfec-fcf4a21a6ead","resolution":{"observed_at":"2026-05-14T23:13:16.105119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T23:31:00.952449Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:19.955943Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:7c6b4a62217025c31ce77cc9774a454e364ff478e33a63c8ec1e584da510400e","observation_id":"6d2bcf41-ade4-4328-9738-7dbb80d785dc","resolution":{"observed_at":"2026-05-11T00:50:50.469751Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-02T16:40:55.445640Z","title":"Multichallenge: A realistic multi-turn conversation evaluation benchmark challenging to frontier llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T23:31:00.952449Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T16:40:55.445640Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:0fc783ecba00facddd2e74cd7d2243a4c32f0653d9baa5bcd863186b4681927c","observation_id":"f6d4d5ec-bc87-4f69-a129-da250de9fe45","resolution":{"observed_at":"2026-08-02T16:40:55.445640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-04T05:36:43.360793Z","title":"Multichallenge: A realistic multi-turn conversation evaluation benchmark challenging to frontier llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T23:31:00.952449Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T05:36:43.360793Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:3bcdea87783c042d96e2e65538c95cd2fcd7cc800f76642b3d29f3066aa0cf52","observation_id":"f49b607a-92f2-41a3-8729-aab61af32184","resolution":{"observed_at":"2026-08-04T05:36:43.360793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2604.16310","last_updated":"2026-01-30T15:32:14Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-30T15:32:14Z","title":"RAG-DIVE: A Dynamic Approach for Multi-Turn Dialogue Evaluation in Retrieval-Augmented Generation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T09:13:55.204404Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2604.16310"},"observation_digest":"sha256:8ffb8f24dfce59129e49adb72d8af04df4c56cfe91b38e0df394d27daac7f58a","observation_id":"c5fb9061-7d35-49af-9232-0e9cad0e51a1","resolution":{"observed_at":"2026-05-16T09:17:40.301640Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2605.27671","last_updated":"2026-05-26T20:48:33Z","snapshot_observed_at":"2026-08-04T03:57:33.093718Z","submitted_at":"2026-05-26T20:48:33Z","title":"Evolving and Detecting Multi-Turn Deception using Geometric Signatures","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T15:17:58.804973Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2605.27671"},"observation_digest":"sha256:a9724952d2d90cc50bf712aa29c4fa5c71600cdc26b326af46d2e21601f2d107","observation_id":"2983f954-59cb-4294-81c6-ac074e7e8e25","resolution":{"observed_at":"2026-06-29T15:23:32.697954Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2606.09878","last_updated":"2026-06-03T01:28:00Z","snapshot_observed_at":"2026-07-06T23:49:08.412079Z","submitted_at":"2026-06-03T01:28:00Z","title":"FailureScope: Cross-Regime Behavioral Diagnosis of Language Model Weaknesses","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T07:16:30.725358Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2606.09878"},"observation_digest":"sha256:d8a956ee8b86b260c90be595777cbdf8c86d897f9be99c0e5972b3a03dfdbdd5","observation_id":"9b3ea34e-c4b4-41ff-bf02-b439edf5abf8","resolution":{"observed_at":"2026-07-02T06:56:44.740957Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2606.17114","last_updated":"2026-06-15T09:16:38Z","snapshot_observed_at":"2026-08-02T23:25:19.283217Z","submitted_at":"2026-06-15T09:16:38Z","title":"An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T03:39:36.657903Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2606.17114"},"observation_digest":"sha256:aa28ab63bcdc8390c72a30d39bf35a10f8d856cb0198a7b6acdaf7545464824d","observation_id":"d3050268-e08e-479a-b75d-258cccea1f49","resolution":{"observed_at":"2026-07-03T17:48:46.351978Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":"2501.17399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-07-03T20:48:56.198096Z","title":"arXiv preprint arXiv:2501.17399 , year=","venue":null,"work_id":"a8c4c1d5-2ef5-471b-ac66-d5a37465ebed","year":2023},"citing_paper":{"arxiv_id":"2606.18216","last_updated":"2026-06-16T17:46:02Z","snapshot_observed_at":"2026-07-06T23:53:40.698123Z","submitted_at":"2026-06-16T17:46:02Z","title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","version":1},"reference_index":155,"source":"pdf_text","source_observed_at":"2026-06-27T01:08:52.981296Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2606.18216"},"observation_digest":"sha256:732b3bd6c807119f3db40d88a829b01bb1abeb6f40f3b291330ffe3f81f56e55","observation_id":"36c02bfb-a6e5-40e5-8841-42a8188303c7","resolution":{"observed_at":"2026-07-03T20:48:56.199387Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.17399/citation-record","integrity":"/paper/2501.17399/integrity","json":"/paper/2501.17399/citation-record.json","paper":"/paper/2501.17399"},"outbound":[],"paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-05T22:39:31.964846Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2501.17399."}