{"as_of":"2026-08-15T23:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8e882049da789bbf1a091142be9ff787ac850dff5821f180e6383e6475e3a7d9","coverage":[{"denominator":129,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:08:08.453311Z","state":"measured"},{"denominator":105,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":105,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:16:53.580999Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-15T00:39:35.853384Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-08-15T17:16:53.580999Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.13805","last_updated":"2025-08-19T13:12:01Z","snapshot_observed_at":"2026-08-15T17:08:50.673767Z","submitted_at":"2025-08-19T13:12:01Z","title":"Prompt-Based One-Shot Exact Length-Controlled Generation with LLMs","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-15T17:16:53.580999Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2508.13805"},"observation_digest":"sha256:ea9935b1191e9d1d0a6f76bd53ebf2e928a3c2719f0618e95528ffdca758eedb","observation_id":"6ff77394-c99f-42d7-af36-3730cab6fc15","resolution":{"observed_at":"2026-08-15T17:16:53.580999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":"2505.16234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":"1c993b00-dc5b-4d1f-9fe1-8565a802c08d","year":2025},"citing_paper":{"arxiv_id":"2603.22267","last_updated":"2026-05-13T14:33:44Z","snapshot_observed_at":"2026-08-03T10:51:29.519900Z","submitted_at":"2026-03-23T17:51:40Z","title":"TiCo: Time-Controllable Spoken Dialogue Model","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T00:38:52.182973Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2603.22267"},"observation_digest":"sha256:3b2087979ebbb4413cb0e426fa60bb4355bb9ba34cc6fc8df73684a0ff8af86e","observation_id":"30905453-4f45-4e4b-b4e7-d3288f6a18c7","resolution":{"observed_at":"2026-05-15T00:39:35.854869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":"2505.16234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":"1c993b00-dc5b-4d1f-9fe1-8565a802c08d","year":2025},"citing_paper":{"arxiv_id":"2604.27039","last_updated":"2026-07-20T23:24:02Z","snapshot_observed_at":"2026-08-02T15:18:52.956023Z","submitted_at":"2026-04-29T17:09:21Z","title":"Length Value Model: Scalable Value Pretraining for Token-Level Length Modeling","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-07T10:42:27.644514Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2604.27039"},"observation_digest":"sha256:6ef861c9e2d309455e887415bea86081d9acfd464004447c0e75ae441b1f7b65","observation_id":"eacdae65-3aa9-48a7-84d2-fe4f586cb22c","resolution":{"observed_at":"2026-05-12T09:31:26.140879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-08-02T15:28:14.401581Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.27039","last_updated":"2026-07-20T23:24:02Z","snapshot_observed_at":"2026-08-02T15:18:52.956023Z","submitted_at":"2026-04-29T17:09:21Z","title":"Length Value Model: Scalable Value Pretraining for Token-Level Length Modeling","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T15:28:14.401581Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2604.27039"},"observation_digest":"sha256:826518a8cd80a9e5b7c0a47b6a965d4cd736aace639208c699e0a003f7897c56","observation_id":"463c613c-2a32-44e3-bfe1-3cbabdb90551","resolution":{"observed_at":"2026-08-02T15:28:14.401581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-08-01T18:03:36.857585Z","title":"LIFEBENCH: 16 Evaluating Length Instruction Following in Large Language Models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17420","last_updated":"2026-07-19T21:47:58Z","snapshot_observed_at":"2026-08-15T06:22:54.408639Z","submitted_at":"2026-07-19T21:47:58Z","title":"The Librarian Who Refused to Code: Model-Dependent Identity Enactment in LLM Code Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T18:03:36.857585Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2607.17420"},"observation_digest":"sha256:d2f7837aa0c37a2bb7c1a3be2ab7510581454de2d98a595bca3bf31f5db5924f","observation_id":"9003ed95-f141-480b-8560-443b02ae4816","resolution":{"observed_at":"2026-08-01T18:03:36.857585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.16234/citation-record","integrity":"/paper/2505.16234/integrity","json":"/paper/2505.16234/citation-record.json","paper":"/paper/2505.16234"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:07:59.915175Z","title":"Abedi Firouzjaei","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.915175Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:15886b58b8e38f363cc831ff01a2002486fe2ade7060f79613fc143d9f3f1b8e","observation_id":"d03fcbf4-d152-4577-8b0c-e084be42cdbe","resolution":{"observed_at":"2026-08-07T15:07:59.915175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:07:59.980923Z","title":"Alzantot, Y","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.980923Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:1b62433372fbdd4b9d996ac3de63d97cc113539ca872a91beac33bc41388a5c4","observation_id":"52722d46-9a3d-4a13-9c58-5dec7b1fdc4a","resolution":{"observed_at":"2026-08-07T15:07:59.980923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.100922Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.100922Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:547b4466c19bc5b992e231c16d4b286c3145ca81ae4bb824fd2f22a36d3d9eba","observation_id":"d62c8ae7-3dfe-454f-bb53-6b6025bae52f","resolution":{"observed_at":"2026-08-07T15:08:00.100922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.243342Z","title":"Claude 3.7 Sonnet and Claude Code","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.243342Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cb05e27f7d955a5fc3ee9ffb2c3f968ec0cf8ea80a8c22127b855f60a50bfaac","observation_id":"12609bb8-f46c-4f25-88c3-f655ffd5fb18","resolution":{"observed_at":"2026-08-07T15:08:00.243342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.370074Z","title":null,"venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.370074Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a647c528179eda3f13332c58ed2c7b50ace9e03d40f552b47780382a5325932a","observation_id":"75e7b2b1-f57d-478d-8d66-009a1e343575","resolution":{"observed_at":"2026-08-07T15:08:00.370074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.482208Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.482208Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f33d51de5efdfeba98ca2c5b25eaf672079df362d5be0846ae77a8c513202f13","observation_id":"76e6cf04-00d6-4840-bdfa-52c5db7dd6ca","resolution":{"observed_at":"2026-08-07T15:08:00.482208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15204","last_updated":"2025-01-03T11:44:51Z","snapshot_observed_at":"2026-08-12T12:34:50.226758Z","submitted_at":"2024-12-19T18:59:17Z","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15204","snapshot_observed_at":"2026-08-07T15:08:00.616769Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.616769Z"},"links":{"cited_paper":"/paper/2412.15204","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:239bad8568fc2f848ba975bc93b9a40aae839f74e15328a27323f431fa06914a","observation_id":"f427873a-8f62-4dc0-91be-8d9997e3035f","resolution":{"observed_at":"2026-08-07T15:08:00.616769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.759419Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.759419Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:bf730b19789560e2f4ec77d6702456fee2e6bb6b18d35b344048c847a5406424","observation_id":"c8073588-1d7d-4018-a79e-6ede488102e6","resolution":{"observed_at":"2026-08-07T15:08:00.759419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.923158Z","title":"Bordes, Y .-L","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.923158Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:3732b0f21b4cad3e38c9fb29c288c555d4300c08d353bb80ad25ed6089afbff1","observation_id":"85558c24-4962-4f40-9ab3-7f563662fba8","resolution":{"observed_at":"2026-08-07T15:08:00.923158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.022186Z","title":"Bosselut, A","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.022186Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:0460194df2b3b4bd9150206d9f2f9983432edecb6fed33ac3943c087b16a8f93","observation_id":"8a000c4b-be89-4539-878b-9cbd7123646c","resolution":{"observed_at":"2026-08-07T15:08:01.022186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.144208Z","title":"Butcher, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.144208Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:7cf2c92901501f18e64113be989cd2ec8d4f065da99a57fd2cd3bf0e3c75d510","observation_id":"dee3ee11-3c31-4520-b503-0d57e76e40e9","resolution":{"observed_at":"2026-08-07T15:08:01.144208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.233200Z","title":"Doubao-1.5-Pro","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.233200Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:933bf1ee5baa26d817ee494a5dc7fc33a59aa9901230f79c9e652cec038d9bf9","observation_id":"0cec495e-48db-4aa0-aa49-185bd1bb0555","resolution":{"observed_at":"2026-08-07T15:08:01.233200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.312710Z","title":"Chang, X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.312710Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:8fc5c08e0c218cdfa296b1caa1deb7cc23e0e1cef2de83b947be330e4ef6434a","observation_id":"d83dcc59-f853-4d12-ad7f-9fb3c1bf25dc","resolution":{"observed_at":"2026-08-07T15:08:01.312710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T15:08:01.389348Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.389348Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:0345fca31c00f41d8e452ad83785d538d2cb06dd75773c469c7b407755793e1f","observation_id":"2845bc06-2ac8-469d-a358-83afe998016c","resolution":{"observed_at":"2026-08-07T15:08:01.389348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.475747Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.475747Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:676e1ae93f43028d4e87f68be355f5e739783eb0f373e3e84eab7ba42d394079","observation_id":"3a2e6bd6-f44e-4228-b0da-817dd6009ed0","resolution":{"observed_at":"2026-08-07T15:08:01.475747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.576145Z","title":"Chen and C","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.576145Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9cd2fe066d410d63835b260810e602afd12777c91c18b2ee87622abdeeea329d","observation_id":"1b9c43a9-c31c-4586-b9f3-67543687f7d8","resolution":{"observed_at":"2026-08-07T15:08:01.576145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.643294Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.643294Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ae3d917abde8633fb1423506e9ad6685f11eef0bd641cd24f15394bee2d7f919","observation_id":"7b252921-98f3-457a-b125-ee69f5ac761e","resolution":{"observed_at":"2026-08-07T15:08:01.643294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.723680Z","title":"Chiang, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.723680Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b87758957539d909d1eea72f637e7c0fe30276462404793803d002377977c4f9","observation_id":"f5bcecf5-7db7-4429-9b1b-7a332a67e84a","resolution":{"observed_at":"2026-08-07T15:08:01.723680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.821582Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.821582Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6567aa1fd83859765ef9f0254d325ed47e123b26e3b398a69f11bdb68c65cf93","observation_id":"9f64694b-b587-4229-9e0e-87e2b99537da","resolution":{"observed_at":"2026-08-07T15:08:01.821582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T15:08:01.907072Z","title":"Cobbe, V","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.907072Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:db4f193401d61886e59a7ee34a9605181cd51a1e43f9c08e4c2929547c0ed4f4","observation_id":"9ee681a9-1b69-4f20-84b7-9353dea5e07a","resolution":{"observed_at":"2026-08-07T15:08:01.907072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.090683Z","title":"Cohan, F","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.090683Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d94813f7d659016c1d67ea3451399f3ca8cabc19d86c983d0b70c57c10a41e82","observation_id":"5eef2f6c-3b9a-4e35-bcc9-6219b1c0f537","resolution":{"observed_at":"2026-08-07T15:08:02.090683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.249374Z","title":"Collobert, J","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.249374Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ab9e0c8eaf946aaea4a13512c164f420a561bf2779ca7eb2f3d434017f220309","observation_id":"b8a754b4-d26d-4973-8e81-433936305d43","resolution":{"observed_at":"2026-08-07T15:08:02.249374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08268","last_updated":"2025-07-09T17:25:55Z","snapshot_observed_at":"2026-08-15T22:44:46.432581Z","submitted_at":"2024-12-11T10:35:45Z","title":"LCFO: Long Context and Long Form Output Dataset and Benchmarking","version":3},"cited_work":{"arxiv_id":"2412.08268","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.08268","snapshot_observed_at":"2026-08-07T15:08:12.041708Z","title":"LCFO: Long Context and Long Form Output Dataset and Benchmarking","venue":"cs.CL","work_id":"a570e7c4-1c95-4972-8f96-7f4d735aefdb","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.336424Z"},"links":{"cited_paper":"/paper/2412.08268","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4b0f854f015e8df98e1df038ec62b26e0e232e85a23657756635fe9f4e1e9d10","observation_id":"642495c2-3a0b-4a7a-9b68-31d725882625","resolution":{"observed_at":"2026-08-07T15:08:12.116541Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.496551Z","title":"Davidson, D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.496551Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b02e668eca6c4c1101509f961dc795973ab0b4d725b631332f1e2e85032a37a7","observation_id":"8575674f-54ee-4641-8746-f565c54fe5dc","resolution":{"observed_at":"2026-08-07T15:08:02.496551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.698130Z","title":null,"venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.698130Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d057dbf09470b95c411420482cbfc41d23ebd7c190d6774b0aa10c1ba53c21bc","observation_id":"264207e3-a4f7-49e7-9fea-48f19a70c967","resolution":{"observed_at":"2026-08-07T15:08:02.698130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.802654Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.802654Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cb096dc94bfd73834bc8f540928c28cf233fa7331985373e482266b019bdeb71","observation_id":"207f2d8c-d40f-4223-8f31-f95caddd5bb1","resolution":{"observed_at":"2026-08-07T15:08:02.802654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.954068Z","title":"Dubois, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.954068Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f5a4b540a38c0e4e642e53c0cdf54aa18037a8d66e19f5ea438d45fe98841b6c","observation_id":"348202ab-4617-46ca-b372-c9179b218782","resolution":{"observed_at":"2026-08-07T15:08:02.954068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.106742Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.106742Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4795257461b074dbef6e8a1deb8c119c35f5b3ad90fe6872fac15f0887fc85bc","observation_id":"74eced30-2b84-4219-8b65-3311a6304d66","resolution":{"observed_at":"2026-08-07T15:08:03.106742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.246808Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.246808Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:59313ddec7f6b4177dee07a7b4a1b5b7f2664d1a0d2fde24bc0cd0129c89439f","observation_id":"29979e30-9658-4795-aae9-fb2f03def193","resolution":{"observed_at":"2026-08-07T15:08:03.246808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.356120Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.356120Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d52d3f631ce90de061b179b12ab95ed250b458e5745332f45ff82cdadda03bd3","observation_id":"02d05053-243d-4d2a-9694-2ef27cfb62d8","resolution":{"observed_at":"2026-08-07T15:08:03.356120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.507635Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.507635Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:82e6f0fad53552976ec5473f20f4b64c6a864d57b3d779c144a565e39ffb0954","observation_id":"7520112c-1822-41e5-aeb0-119846f48bbb","resolution":{"observed_at":"2026-08-07T15:08:03.507635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.619331Z","title":"Foundation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.619331Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e3067666238b09897b71c39e05a7f8dcd7562a0f906aec84606726eb67781d28","observation_id":"1f012298-b49c-4271-9aa4-65d5d67af0c8","resolution":{"observed_at":"2026-08-07T15:08:03.619331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-14T09:56:00.692687Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-07T15:08:03.761610Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.761610Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ccf1b24f7abc7d749e169261cd6568473c2ecbf1f388ca95bf905cef6a53f93b","observation_id":"c4b3fe12-b0fc-4900-b9c7-ee56d1e7a38c","resolution":{"observed_at":"2026-08-07T15:08:03.761610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.821979Z","title":"Gemini 2.0 Flash","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.821979Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b9a8454db8cdb31ae529fbf041d6d1c54c4493dd742c003375962e40d0063e7b","observation_id":"31377b83-3942-4967-b76e-687b8a78e71f","resolution":{"observed_at":"2026-08-07T15:08:03.821979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.877823Z","title":"Gemini 2.5 Pro","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.877823Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b8b0003a18256f7f139725aa3acb50c323bdc5fc5d363884a5fe2959fec4defe","observation_id":"ed19b53f-3d7b-4290-8c8b-879e4119f478","resolution":{"observed_at":"2026-08-07T15:08:03.877823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T15:08:03.923252Z","title":"Grattafiori, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.923252Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a7317bed871c3059b20c36ca57dfc1fe35cca3cf63d3b83a11af31c08667e480","observation_id":"9e937a04-a837-4446-a22f-80a85be20f3b","resolution":{"observed_at":"2026-08-07T15:08:03.923252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.998056Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.998056Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:14dd78b1756deaa9cb74590e8a6f7d24b09caa511416d0f289ceaf29c2cb4424","observation_id":"f19c070b-17d7-4cfa-9a32-0a7e8985ddd5","resolution":{"observed_at":"2026-08-07T15:08:03.998056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-13T06:43:22.011336Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-07T15:08:04.044070Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.044070Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9370a407b6e2b3cffc90619850a4fac4fb56d341bcc32e40f019cd688a28052b","observation_id":"caef362d-1e6e-4407-81dc-227fe3ba992a","resolution":{"observed_at":"2026-08-07T15:08:04.044070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14656","last_updated":"2024-12-19T09:07:38Z","snapshot_observed_at":"2026-08-13T15:12:29.254750Z","submitted_at":"2024-12-19T09:07:38Z","title":"Length Controlled Generation for Black-box LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14656","snapshot_observed_at":"2026-08-07T15:08:04.110426Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.110426Z"},"links":{"cited_paper":"/paper/2412.14656","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6f45a7b07e14209a4f0b775ddf28dfb81b2cd1264f4c72fe494fe96d43a21984","observation_id":"0a1bbbcf-23ea-4ed9-9ec2-9ab31402e39f","resolution":{"observed_at":"2026-08-07T15:08:04.110426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:08:04.157989Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.157989Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c61285723e26e5f57168ba7a839a92b60fd2698abf56f9943a3069401607b400","observation_id":"6e458285-5368-4628-a9a7-9cfa31d92340","resolution":{"observed_at":"2026-08-07T15:08:04.157989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15553","last_updated":"2024-11-13T04:26:13Z","snapshot_observed_at":"2026-08-12T22:19:04.052792Z","submitted_at":"2024-10-21T00:59:47Z","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15553","snapshot_observed_at":"2026-08-07T15:08:04.207512Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.207512Z"},"links":{"cited_paper":"/paper/2410.15553","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ab1760faf130dab9380cb3b057fd5e92de7b3749453458d484883e3ef6702b6d","observation_id":"f3ab0426-6758-4aac-bd61-241b9d68ca63","resolution":{"observed_at":"2026-08-07T15:08:04.207512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.244676Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.244676Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:07b9a1b9b2c79a61c56a4fb0852ad5d2e66d6e80823bb5b3924e89c2e57235a3","observation_id":"fd5c61b4-e6f6-45aa-8607-9abb2af7ac50","resolution":{"observed_at":"2026-08-07T15:08:04.244676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.314289Z","title":"Hsieh, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.314289Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6719cd7d8ec1bf6e2ab6b39638ac7ecf0dcd235026bdf812090b2da9bb26123f","observation_id":"48bc6929-6b5d-479b-b24e-28ae7ef193cd","resolution":{"observed_at":"2026-08-07T15:08:04.314289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.379241Z","title":"Huang and K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.379241Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5430412f83e98900eb8a87990c63227ba03f52864e67639af2da0f249f20f387","observation_id":"0e770906-3b15-4109-8d83-d733cff41d74","resolution":{"observed_at":"2026-08-07T15:08:04.379241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.438908Z","title":"Huang, X","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.438908Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:74dc77828d9ff4be35571f11a492e44fd7a65b4f803bcd2d8ffcf18c92bd5ad1","observation_id":"eaee9ae1-cf63-48cb-84a5-8bfc7eedc978","resolution":{"observed_at":"2026-08-07T15:08:04.438908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15777","last_updated":"2024-05-29T15:50:43Z","snapshot_observed_at":"2026-08-15T15:12:39.657853Z","submitted_at":"2024-04-24T09:55:24Z","title":"A Comprehensive Survey on Evaluating Large Language Model Applications in the Medical Industry","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15777","snapshot_observed_at":"2026-08-07T15:08:04.497560Z","title":"Huang, K","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.497560Z"},"links":{"cited_paper":"/paper/2404.15777","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ac1c67188699fba670ab1e6b30c50454dfb159178446ea02dda2c71ecba771f4","observation_id":"5bcac5cd-9a07-4a6b-92fc-abaa0f1faeae","resolution":{"observed_at":"2026-08-07T15:08:04.497560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03200","last_updated":"2025-01-06T18:28:04Z","snapshot_observed_at":"2026-08-10T21:50:40.923870Z","submitted_at":"2025-01-06T18:28:04Z","title":"The FACTS Grounding Leaderboard: Benchmarking LLMs' Ability to Ground Responses to Long-Form Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03200","snapshot_observed_at":"2026-08-07T15:08:04.558589Z","title":"Jacovi, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.558589Z"},"links":{"cited_paper":"/paper/2501.03200","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f676aa8cf4ba2aae01381476e24c294d479286288821b81e4471a60da1f9ea83","observation_id":"9b0ac765-2a37-4dd1-9ad1-3c987da4f752","resolution":{"observed_at":"2026-08-07T15:08:04.558589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T15:08:04.604933Z","title":"Jaech, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.604933Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:57cd993e00fa6d990f2707b99401a25440b304becb905ba205c84379734ba1d8","observation_id":"5f1ff18e-cd0c-4fee-8eae-5582c8694fa2","resolution":{"observed_at":"2026-08-07T15:08:04.604933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.656826Z","title":"Jhamtani, V","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.656826Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:1c110b43247ef0f714577083f87398fd01794760b51766432a939d7cb095a29a","observation_id":"67340472-1885-4a2d-8bcb-ab51bfa796ae","resolution":{"observed_at":"2026-08-07T15:08:04.656826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.700834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.700834Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:796169b84d12247fc3545535d5372e90c6836ece843ff7fe234e85d1ad5a2d8c","observation_id":"4201875f-84b9-4362-abea-85aa730b65fb","resolution":{"observed_at":"2026-08-07T15:08:04.700834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.773018Z","title":"webnovel_cn (revision 745338c), 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.773018Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c16b616dc238533f32238a19ce92f3436aadeac039a95b478d28f735178b8de8","observation_id":"f8f43d67-01fe-4215-b405-3477faba6465","resolution":{"observed_at":"2026-08-07T15:08:04.773018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.828592Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.828592Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:98c93d4e04b5ce8c05848f061f3fa98f23d933c2cf65e1c8afa4ab753ddc9f4d","observation_id":"01265994-1f64-48b2-b6cb-afa6fad5df13","resolution":{"observed_at":"2026-08-07T15:08:04.828592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.891231Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.891231Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4dc9adff01b39155fd61d6560d0f366b1f4daddab370c69274e13360315ec396","observation_id":"57ab9db9-b0ff-480c-b1f8-5985cdbab959","resolution":{"observed_at":"2026-08-07T15:08:04.891231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.954876Z","title":"Koupaee and W","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.954876Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:322f592a71fc5b067517fcc30b45a3118f687fc59907074eb30f7cdc4d5a4c3c","observation_id":"b05c97b9-1322-40df-8b17-02e580e2fcc5","resolution":{"observed_at":"2026-08-07T15:08:04.954876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.023199Z","title":"Kry´sci´nski, N","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.023199Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:1ba4dabf8a581c9990d9a55335285a04b883fc8ed6fb3010f2ac533c744e54bc","observation_id":"0a4ae45a-d174-4dbe-82ed-da489038619a","resolution":{"observed_at":"2026-08-07T15:08:05.023199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.089583Z","title":"Kuratov, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.089583Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a79cbece90e699466d6d4ab713940ad9a5250765b1c71ea4db2b85bece5762dd","observation_id":"2ce96cb9-3ce3-4822-8fd9-f5373e7b8f16","resolution":{"observed_at":"2026-08-07T15:08:05.089583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.137728Z","title":"Lample, M","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.137728Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f04654949c79e11f0d56b2a507fbfc6d59e28aafd3a2cd78707e4ec6dd5f1532","observation_id":"06ebd402-d64f-402d-a331-cf01a447796f","resolution":{"observed_at":"2026-08-07T15:08:05.137728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.197847Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.197847Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ecc77cb8a2470310e1f6a5e796ac811ce5f322a1144fa8c9bcad60d0b80ee210","observation_id":"f5181dd5-d25f-48fb-83d8-73d647c30973","resolution":{"observed_at":"2026-08-07T15:08:05.197847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02060","last_updated":"2024-06-12T02:46:16Z","snapshot_observed_at":"2026-08-15T14:57:42.705501Z","submitted_at":"2024-04-02T15:59:11Z","title":"Long-context LLMs Struggle with Long In-context Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02060","snapshot_observed_at":"2026-08-07T15:08:05.259340Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.259340Z"},"links":{"cited_paper":"/paper/2404.02060","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e5aeb86c84ffc07ef581903eed5250fc4cd4610837ef1c7b594571ed982c3acb","observation_id":"49645e82-b652-4441-9933-cf3cd5d6f730","resolution":{"observed_at":"2026-08-07T15:08:05.259340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20084","last_updated":"2025-06-29T16:03:19Z","snapshot_observed_at":"2026-08-09T12:35:23.283697Z","submitted_at":"2025-04-25T16:03:50Z","title":"AI Awareness","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20084","snapshot_observed_at":"2026-08-07T15:08:05.339186Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.339186Z"},"links":{"cited_paper":"/paper/2504.20084","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d14473f9e9539bbf630b88f34cd9450aa9e6d418804522afb8f3736c2bf80102","observation_id":"9555014a-566b-40ee-927f-818b14f6d62b","resolution":{"observed_at":"2026-08-07T15:08:05.339186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.453724Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.453724Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:613374e00a6ef660faa4db1fc0939513a4ca329d1d4567b14f2fed9aeb6178cf","observation_id":"24f626ea-adfb-44ea-94c5-37195056c79d","resolution":{"observed_at":"2026-08-07T15:08:05.453724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12599","last_updated":"2024-08-22T17:59:04Z","snapshot_observed_at":"2026-08-12T22:59:04.689984Z","submitted_at":"2024-08-22T17:59:04Z","title":"Controllable Text Generation for Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12599","snapshot_observed_at":"2026-08-07T15:08:05.544649Z","title":"Liang, H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.544649Z"},"links":{"cited_paper":"/paper/2408.12599","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:3ab0bd4abb698354144cb1470142a2c09c091b6143b8a8251f7ec53675122504","observation_id":"4d97b2c8-357a-470f-8ada-2521b0871e72","resolution":{"observed_at":"2026-08-07T15:08:05.544649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.644559Z","title":"Lightman, V","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.644559Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e85e771a22317a9056981fc004f5a298f659e4e0ec1c28f199ec4937fa90b651","observation_id":"730f358e-1ce1-4107-bdd1-0a7bb8fc9a1c","resolution":{"observed_at":"2026-08-07T15:08:05.644559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.718550Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.718550Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5ea8886fc2c191996df8b8f8138126aeffc7681ca4d5c6e60b6d3de92102aff1","observation_id":"bda9e038-2d63-4e11-90c4-3171e0cd5aa6","resolution":{"observed_at":"2026-08-07T15:08:05.718550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.827379Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.827379Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f80959e8eea1d840e11040af25c17ca3cb6db556876b8c506e977efc39ff8e5e","observation_id":"2fe3fbff-a7f4-4f7c-a39c-f29a07718948","resolution":{"observed_at":"2026-08-07T15:08:05.827379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-15T17:27:11.980940Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T15:08:05.916363Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.916363Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e7e2b961b2a65777e00312f9f282b0177448407d70665efe10449abf64df5318","observation_id":"9f8e5921-e614-4fe2-9239-c1d4d2eca7e8","resolution":{"observed_at":"2026-08-07T15:08:05.916363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.968830Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.968830Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:875c8b3b2b33d026fa20cc03df50b6e454ef9d2b8c21fbb2aa6ce640af2d88b0","observation_id":"cd80fcd9-43ab-461a-9333-a5737480603a","resolution":{"observed_at":"2026-08-07T15:08:05.968830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:06.040972Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.040972Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:11547a148ee6147e058f5c6dbfad9946b9e790a12889771fcd93e266449c8dc1","observation_id":"a8ce4bf1-95c6-4c87-aaa5-dc73ae8255f3","resolution":{"observed_at":"2026-08-07T15:08:06.040972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.920395Z","title":null,"venue":null,"work_id":"ae3877d7-c62e-484c-9c32-efc2be9cc961","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.138421Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:862be002de4a9372c2a4b314062216632ec69cf80b5ee78dd749ea36b7ce30e1","observation_id":"f44b67bb-10ee-43a8-ac99-29f04933732a","resolution":{"observed_at":"2026-08-07T15:08:20.965757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:06.237974Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.237974Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:989fbafbdcb6ae6cfa22f5f4a296482eb104e0b5b15e84d188fb57c01aa52067","observation_id":"fe4da9eb-02ca-4ea7-9620-ebff6fd0f76e","resolution":{"observed_at":"2026-08-07T15:08:06.237974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07852","last_updated":"2024-04-02T01:07:05Z","snapshot_observed_at":"2026-08-13T10:13:31.085157Z","submitted_at":"2023-09-14T16:54:34Z","title":"ExpertQA: Expert-Curated Questions and Attributed Answers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07852","snapshot_observed_at":"2026-08-07T15:08:06.311393Z","title":"Malaviya, S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.311393Z"},"links":{"cited_paper":"/paper/2309.07852","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:0d23f34ccdc7f87d53891b4f6108509811aac05a1e9afa05193daffef720e014","observation_id":"5380913f-d36a-42ef-b2c7-db46a4ade9d3","resolution":{"observed_at":"2026-08-07T15:08:06.311393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.764924Z","title":null,"venue":null,"work_id":"541adb6f-7058-404a-a158-5287163b5823","year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.397371Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:859f600bfca895bae853d5435ea47fbd5c967945dbb48ab8756ed804057c7953","observation_id":"76fa2654-0be0-4af4-8c2c-89fa5fe9345d","resolution":{"observed_at":"2026-08-07T15:08:20.823549Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.725586Z","title":"Chinesenlpcorpus","venue":null,"work_id":"4c7cfa85-7605-4ff8-8047-28a9ef5d0a5c","year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.513962Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:75ac676871097c6a676f54a29767ecd4b28297337f314f29f9748dda434a13ec","observation_id":"69b59cad-1a91-4d25-a19e-dc830b707bc7","resolution":{"observed_at":"2026-08-07T15:08:20.743188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.592029Z","title":"Mnbvc: Massive never-ending bt vast chinese corpus.https://github.com/esbatmop/MNBVC, 2023","venue":null,"work_id":"7af3e35a-e159-4743-ad2a-92cb561ed4af","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.674885Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2c93f50d19a46fbfc851541fd2c63727c89217aa248abd9a6791941cc892e95f","observation_id":"24b62dbc-691b-4772-9348-0ed39bdcd484","resolution":{"observed_at":"2026-08-07T15:08:20.639920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.483502Z","title":"Mostafazadeh, N","venue":null,"work_id":"cf85df4e-0147-4954-b658-325370e4c9e6","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.786681Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:dbd03311ebb1b8aff8be5cdfac635e249acde48ba3264fa8c2dd3c7204dce9e1","observation_id":"1173264d-9c54-422a-93b9-da8a23ce010e","resolution":{"observed_at":"2026-08-07T15:08:20.537826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.283811Z","title":"Nallapati, B","venue":null,"work_id":"f5701cb4-2a73-469a-8aba-af4c713e1ef7","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.882313Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9ad3cc89ea29c766cafef053a3e79eb13e61879fbba4fb023b7cd9b0679a5d96","observation_id":"85b58e44-50d7-4733-8ca1-173a4139db81","resolution":{"observed_at":"2026-08-07T15:08:20.369749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.123437Z","title":"GPT-4o mini: advancing cost-efficient intelligence","venue":null,"work_id":"d1e6b443-eaf6-4dee-b1dd-37ade7ea2405","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.933239Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:57d29d179d928a99ea3796acea7175f3c352ca0735c6278fedf44365dee6e7c9","observation_id":"f6c223a5-af30-4fcf-a99f-2bbb5af790b9","resolution":{"observed_at":"2026-08-07T15:08:20.180851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:07.032956Z","title":"Hello GPT-4o.https://openai.com/index/hello-gpt-4o/, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.032956Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d9037ab26fa686acd0a7776f7040591d7f5cae96506096831e17cc98cc3f8f18","observation_id":"0dd40658-09f2-4836-9104-014091f5039e","resolution":{"observed_at":"2026-08-07T15:08:07.032956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.909602Z","title":"OpenAI o1-mini: Advancing cost-efficient reasoning","venue":null,"work_id":"4d40e9fb-4bb4-41fa-a337-afa3c0abbf35","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.136231Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5662b5e0b8a305341a0a6c5ea227d94b38bef9d293ce05935826fa97ee5bd41d","observation_id":"c04d3ce7-0fc1-4c56-890c-8de0610a4e57","resolution":{"observed_at":"2026-08-07T15:08:20.018525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.674829Z","title":"OpenAI o3-mini: Pushing the frontier of cost-effective reasoning","venue":null,"work_id":"d5c85db0-70e8-4967-a2a7-ab86ed05c9d0","year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.212823Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c2ce984fb8f172a68fcfa6f004be5948426109dcfb6157103c3849f241f80cf3","observation_id":"a5e6597d-4497-42ea-8a4e-8e43c9b6123b","resolution":{"observed_at":"2026-08-07T15:08:19.781683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.477592Z","title":null,"venue":null,"work_id":"0fe7da37-c07b-4519-9192-7470aba49654","year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.302010Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6213d7ed165d270397ac79ff116211fff7102a7b9fe0a4a073290940b3661bed","observation_id":"e2b1c336-0ac1-4c52-8c46-6131af0bed4a","resolution":{"observed_at":"2026-08-07T15:08:19.563974Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.326209Z","title":null,"venue":null,"work_id":"f0b9b1b7-6f91-4156-8469-83bf21242cf2","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.421996Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ec0d9a4baae416ba0aca46dda4d27ad92219a11e65b60b6930abc7ecd3aa3206","observation_id":"5f1e47e0-eb8d-4b31-9ccb-2af426340b81","resolution":{"observed_at":"2026-08-07T15:08:19.417697Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.202080Z","title":null,"venue":null,"work_id":"44120d1e-2f43-4b79-965b-2d6bcc5aeaad","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.521225Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2ff4179d169a344e55c7a9f8e057bd3877652dbf130a0ad826fb659397e0c825","observation_id":"84f16566-6a20-4804-866e-4ec8307b499e","resolution":{"observed_at":"2026-08-07T15:08:19.248048Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.080746Z","title":null,"venue":null,"work_id":"4a9a53ea-0e0b-4013-957d-88acd1dc44aa","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.583818Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f38d04dbb126ae08ead7cc3417a5a6795fb8c9ca3900b6b23350204f6a0f63f1","observation_id":"5a92f655-6357-4763-aa38-2c13fc4c6f1e","resolution":{"observed_at":"2026-08-07T15:08:19.112561Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23933","last_updated":"2024-10-31T13:47:10Z","snapshot_observed_at":"2026-08-12T22:10:48.930594Z","submitted_at":"2024-10-31T13:47:10Z","title":"Language Models can Self-Lengthen to Generate Long Texts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.23933","snapshot_observed_at":"2026-08-07T15:08:07.651762Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.651762Z"},"links":{"cited_paper":"/paper/2410.23933","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2bf2ce004d6b2f99be67e83352a5aba503344c3bc10f82c5d50889ac21fab650","observation_id":"bb0e9050-1346-4443-b9b1-b22197d61123","resolution":{"observed_at":"2026-08-07T15:08:07.651762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.16191","last_updated":"2024-09-24T15:38:11Z","snapshot_observed_at":"2026-08-12T22:38:31.274535Z","submitted_at":"2024-09-24T15:38:11Z","title":"HelloBench: Evaluating Long Text Generation Capabilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.16191","snapshot_observed_at":"2026-08-07T15:08:07.698689Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.698689Z"},"links":{"cited_paper":"/paper/2409.16191","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:fcfc2d6b236e078edb8fbeb2f96717fa81288b0364f798e07892386843fad73b","observation_id":"27da7725-b122-4352-8b43-795015fecb70","resolution":{"observed_at":"2026-08-07T15:08:07.698689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.933537Z","title":"Radford and K","venue":null,"work_id":"cbf19545-73d7-4f8a-9641-e9b916bb1077","year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.792526Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:8ac84de5feb3aa3ef0e43dffe9039386aee18f193765e62b0342fb7f2877c336","observation_id":"cfa6af53-cd98-4c87-8776-4bde43e9b3ee","resolution":{"observed_at":"2026-08-07T15:08:18.997721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.826205Z","title":"Rafailov, A","venue":null,"work_id":"014c20da-c7c2-4f80-97d9-195c476dab6f","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.833724Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:882430a6604779103eb6e20a7685f1b808bdfb668cca4e4df16299089c4d0c39","observation_id":"ae6662e7-ee89-48b2-be05-5437d31a5453","resolution":{"observed_at":"2026-08-07T15:08:18.862390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.702178Z","title":null,"venue":null,"work_id":"dbd82489-c557-4c97-9813-efc06927660d","year":2015},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.872975Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:32fb740ec1ba81f26fcb25431b390486fe6d3412d582261d6c61e91e0159d17c","observation_id":"91c869c8-9537-4c21-a830-09f45af784a0","resolution":{"observed_at":"2026-08-07T15:08:18.746199Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.581580Z","title":null,"venue":null,"work_id":"b8ea15a9-2243-4eb8-8632-fc31b9a311e6","year":1994},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.912824Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:39d5b51858645409377d5be49b288c2e0d712bf0b19c9a3232d114072a46163d","observation_id":"dd45e7ff-fbd0-4514-b49d-f0ff703a619c","resolution":{"observed_at":"2026-08-07T15:08:18.635380Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.440592Z","title":"Sennrich, B","venue":null,"work_id":"471fb905-0171-4282-9014-fc3ef85fbd99","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.978742Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:05322631a227e32c95983c7710b0603b4941706b1855a28495741b79e28a2450","observation_id":"84318624-9748-4bb5-86af-f06cfb3c21e0","resolution":{"observed_at":"2026-08-07T15:08:18.512717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.281176Z","title":"Shaham, M","venue":null,"work_id":"dbca6183-aaf4-473a-b5a8-e908fc0ff5b8","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.027546Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:db3f1c7c2bdd3093b1ef9cc0b47f32f65f8b4f26d973bcf78e81e0f07207cad0","observation_id":"75dad7df-64c9-439f-ad22-0c65def8afee","resolution":{"observed_at":"2026-08-07T15:08:18.362523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.134419Z","title":"Socher, A","venue":null,"work_id":"0a627e23-9dbe-484d-b106-5474220acf86","year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.073224Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cce24cab03dd2d6231d6b7c0788b28cd6c82550b751493b0c157d6a2a30510fb","observation_id":"bc065065-870c-4664-b9d5-a5dc81738cb4","resolution":{"observed_at":"2026-08-07T15:08:18.204207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.023213Z","title":null,"venue":null,"work_id":"bf6e396a-3509-49f6-a32d-1d2c5a8f74bf","year":2020},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.114897Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4e6afab7dca4951453286be242a6f58152107f49d37295ad9be9fba80257d2bc","observation_id":"d8ffbb68-2cad-46e2-b7a1-28e802e14dff","resolution":{"observed_at":"2026-08-07T15:08:18.065546Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:08.152779Z","title":"Sutskever, O","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.152779Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cfc09689c268041c9151b4a69183dda2157c395c1c9c64eaaddbae810d5bc44e","observation_id":"c2cd78d9-fbae-46fa-9bd9-170b91d2a54a","resolution":{"observed_at":"2026-08-07T15:08:08.152779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.853739Z","title":"Talmor, J","venue":null,"work_id":"0717576b-2db4-49d9-a227-dfc66e499b61","year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.199068Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:913378f9f0baa024efbaa9cdbd350978359d5472f58b2f5c14dba054601c43ab","observation_id":"d8b7c369-252c-4771-ba80-ece7033421e9","resolution":{"observed_at":"2026-08-07T15:08:17.933821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.532461Z","title":null,"venue":null,"work_id":"6449bcaf-0fbf-4dc2-af99-a624e42553e1","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.252432Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ff36f0a9e14ec3848c62f48e60233cd8505d04615f7eb4224810aaa3d4365fd5","observation_id":"aeea2fa9-9cbd-49b9-8125-dfbc067f22bc","resolution":{"observed_at":"2026-08-07T15:08:17.705498Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12665","last_updated":"2025-02-11T02:09:38Z","snapshot_observed_at":"2026-08-14T10:47:14.154500Z","submitted_at":"2024-06-18T14:35:12Z","title":"CollabStory: Multi-LLM Collaborative Story Generation and Authorship Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12665","snapshot_observed_at":"2026-08-07T15:08:08.312320Z","title":"Venkatraman, N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.312320Z"},"links":{"cited_paper":"/paper/2406.12665","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c95f48277490124054308a1f5c05c1910ecd8cff12119f4c51b901447a8dcfc9","observation_id":"4b274721-20f3-4893-9441-212289a8c5cb","resolution":{"observed_at":"2026-08-07T15:08:08.312320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.269639Z","title":null,"venue":null,"work_id":"d1424e90-7ae8-478a-bdf9-d39f0cb7241b","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.393222Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e04a1ad9ee55aaf4b669396c22a7202b918443848c04838effe3035b67af9dc0","observation_id":"fc9de4cb-0846-4cd2-a62e-5f5ab5bb5382","resolution":{"observed_at":"2026-08-07T15:08:17.405637Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15585","last_updated":"2025-06-09T02:36:20Z","snapshot_observed_at":"2026-08-09T14:46:30.842437Z","submitted_at":"2025-04-22T05:02:49Z","title":"A Comprehensive Survey in LLM(-Agent) Full Stack Safety: Data, Training and Deployment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15585","snapshot_observed_at":"2026-08-07T15:08:08.453311Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.453311Z"},"links":{"cited_paper":"/paper/2504.15585","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a7a32fce2d6e55891271796c7eddc9529d0e31b1d079c2a600bbbff85306edcb","observation_id":"617e84f1-45a8-486b-bc09-a595734d39da","resolution":{"observed_at":"2026-08-07T15:08:08.453311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T19:45:00.839614Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":86,"verified_exact":1,"verified_fuzzy":13},"total_outbound_references":129},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 100 of 129 outbound references and 5 inbound Pith citation observations for arXiv:2505.16234."}