{"as_of":"2026-08-10T08:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e496e6aaad1bfb4170aa404df93b9c64e31141c868b0e4cead71a65c008e93a8","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:14:09.828387Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T06:59:19.857066Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T07:04:21.318652Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"cited_work":{"arxiv_id":"2506.14927","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14927","snapshot_observed_at":"2026-06-30T07:04:21.318652Z","title":"Long Phan, Alice Gatti, Nathaniel Li, Adam Khoja, Ryan Kim, and others","venue":null,"work_id":"40277a62-c62d-4f60-bd88-5834fb5d8f9c","year":2026},"citing_paper":{"arxiv_id":"2606.29630","last_updated":"2026-06-28T22:27:26Z","snapshot_observed_at":"2026-08-03T22:42:04.127016Z","submitted_at":"2026-06-28T22:27:26Z","title":"SFBench: The SciFy Scientific Feasibility Benchmark","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T06:59:19.857066Z"},"links":{"cited_paper":"/paper/2506.14927","citing_paper":"/paper/2606.29630"},"observation_digest":"sha256:4f03a3a40c85ce3ff6cee3b97314bb8e98916edf7ad41cb25548fad4d21cb366","observation_id":"3536efbd-6f6c-4e6c-a7ba-6c21a06588bc","resolution":{"observed_at":"2026-06-30T07:04:21.320199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.14927/citation-record","integrity":"/paper/2506.14927/integrity","json":"/paper/2506.14927/citation-record.json","paper":"/paper/2506.14927"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:12.150651Z","title":null,"venue":null,"work_id":"b9d0fcaf-4aa2-4497-a84a-1418eabeaece","year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:04.588951Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:555ac3cb61b0f10fab4fed3388de66c73d64e195e4f37cc4b0f238642931d3fc","observation_id":"84642ad4-dd72-43c7-8745-1f5b96519f04","resolution":{"observed_at":"2026-08-07T00:14:12.197914Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00406","last_updated":"2021-09-02T23:46:38Z","snapshot_observed_at":"2026-08-07T22:44:06.053099Z","submitted_at":"2021-01-02T09:01:39Z","title":"CDLM: Cross-Document Language Modeling","version":2},"cited_work":{"arxiv_id":"2101.00406","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.00406","snapshot_observed_at":"2026-08-07T00:14:11.885779Z","title":"CDLM: Cross-Document Language Modeling","venue":"cs.CL","work_id":"6e1d5ec0-d0b2-4896-940e-bac50d083bb1","year":2021},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:04.692123Z"},"links":{"cited_paper":"/paper/2101.00406","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:29773b0e1474ddc18781c3ddd22f0baeb7a4f427eb1a520e2e59723a46a3e39e","observation_id":"a42575fb-443f-4959-8981-56068807a269","resolution":{"observed_at":"2026-08-07T00:14:12.019518Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:04.837567Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:04.837567Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:0c9e086a133afb76bef651ac4ee3a777610498066740f6826ed639883235d4ee","observation_id":"dfe629bc-9b00-4c72-ab71-cd4411504d28","resolution":{"observed_at":"2026-08-07T00:14:04.837567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.02164","last_updated":"2020-06-14T19:14:22Z","snapshot_observed_at":"2026-08-07T22:43:28.600835Z","submitted_at":"2019-09-05T00:25:17Z","title":"TabFact: A Large-scale Dataset for Table-based Fact Verification","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.02164","snapshot_observed_at":"2026-08-07T00:14:05.014463Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:05.014463Z"},"links":{"cited_paper":"/paper/1909.02164","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:ac1ed0a076200a8a006504a2eab4a150d9c5ea1da896ebe96535c6f5e81fb098","observation_id":"cb60734e-0938-422d-9cbb-a2c85ccafb43","resolution":{"observed_at":"2026-08-07T00:14:05.014463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T00:14:05.920857Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:05.920857Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:b81ca6a3bf047f0d6b16eda85c98db03a1da3485c593ed271e856e550d1c0fdd","observation_id":"b4b7eacd-90db-4f9e-8970-866bc4ab3717","resolution":{"observed_at":"2026-08-07T00:14:05.920857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:06.028201Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.028201Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:8c1ed35a757e1a5889a11f312cb90850ba4e449899907996826fb67565a78315","observation_id":"afa6ec3e-d9d7-4181-af4f-95c47d58ff35","resolution":{"observed_at":"2026-08-07T00:14:06.028201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.07934","last_updated":"2023-06-13T17:39:20Z","snapshot_observed_at":"2026-08-07T22:45:12.985098Z","submitted_at":"2023-06-13T17:39:20Z","title":"BoardgameQA: A Dataset for Natural Language Reasoning with Contradictory Information","version":1},"cited_work":{"arxiv_id":"2306.07934","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.07934","snapshot_observed_at":"2026-08-07T00:14:11.609724Z","title":"BoardgameQA: A Dataset for Natural Language Reasoning with Contradictory Information","venue":"cs.CL","work_id":"2b4bea81-4452-4324-bb9b-50f5c4f5b29c","year":2023},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.176121Z"},"links":{"cited_paper":"/paper/2306.07934","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:18e244a707938bf98f685224e6a65a169215fb22f47102401004cee1afae3050","observation_id":"db67d57c-c146-45f2-a605-53a012384eb8","resolution":{"observed_at":"2026-08-07T00:14:11.683263Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12447","last_updated":"2024-08-09T12:40:51Z","snapshot_observed_at":"2026-08-09T15:41:46.050587Z","submitted_at":"2024-04-18T18:12:01Z","title":"AmbigDocs: Reasoning across Documents on Different Entities under the Same Name","version":3},"cited_work":{"arxiv_id":"2404.12447","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.12447","snapshot_observed_at":"2026-08-07T00:14:11.413141Z","title":"AmbigDocs: Reasoning across Documents on Different Entities under the Same Name","venue":"cs.CL","work_id":"3f5e3b82-5501-43d5-b2ef-354af918ced9","year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.307677Z"},"links":{"cited_paper":"/paper/2404.12447","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:11b7b26286f0462f542a66896064009f8addc5cde63e6116565e03ec949a7679","observation_id":"249ac8a1-fadb-4ea4-8874-c7d9e75ef9dd","resolution":{"observed_at":"2026-08-07T00:14:11.514239Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.19308","last_updated":"2023-10-30T06:36:24Z","snapshot_observed_at":"2026-07-06T15:35:33.952097Z","submitted_at":"2023-05-30T17:59:30Z","title":"SheetCopilot: Bringing Software Productivity to the Next Level through Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.19308","snapshot_observed_at":"2026-08-07T00:14:06.425652Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.425652Z"},"links":{"cited_paper":"/paper/2305.19308","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:e9f1bcc35945b44bf702b5d0e646dc20c81921bc4d977fbd0ff0e741a71b19fe","observation_id":"553757b3-5fa4-46ac-86bf-9b6d2eb1c25a","resolution":{"observed_at":"2026-08-07T00:14:06.425652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:06.569954Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.569954Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:a21260ed407c38459a604aed82db96207f4ee740a5a40be057999ba656044de1","observation_id":"80f743c7-a0c9-425e-8078-91308b831594","resolution":{"observed_at":"2026-08-07T00:14:06.569954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16086","last_updated":"2024-06-23T11:57:53Z","snapshot_observed_at":"2026-08-07T06:46:02.082943Z","submitted_at":"2024-06-23T11:57:53Z","title":"SEAM: A Stochastic Benchmark for Multi-Document Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16086","snapshot_observed_at":"2026-08-07T00:14:06.720297Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.720297Z"},"links":{"cited_paper":"/paper/2406.16086","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:7666ce6499bc56287260b581ba36bab66ab1358ae484e39d5859364f387813f0","observation_id":"d860a126-1498-4a96-b73a-f519947c2919","resolution":{"observed_at":"2026-08-07T00:14:06.720297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07503","last_updated":"2024-08-10T20:46:47Z","snapshot_observed_at":"2026-07-06T17:58:39.485508Z","submitted_at":"2024-04-11T06:34:17Z","title":"Best Practices and Lessons Learned on Synthetic Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07503","snapshot_observed_at":"2026-08-07T00:14:06.839189Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.839189Z"},"links":{"cited_paper":"/paper/2404.07503","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:a0d2d168dbc208f49adee5ce4ad8cb5c499beee58d382d39dd0423bcfb7337fe","observation_id":"fd994a8a-6e91-4559-84f9-c47e756904ad","resolution":{"observed_at":"2026-08-07T00:14:06.839189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15126","last_updated":"2024-06-14T07:47:09Z","snapshot_observed_at":"2026-08-09T02:49:33.324683Z","submitted_at":"2024-06-14T07:47:09Z","title":"On LLMs-Driven Synthetic Data Generation, Curation, and Evaluation: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15126","snapshot_observed_at":"2026-08-07T00:14:06.967525Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:06.967525Z"},"links":{"cited_paper":"/paper/2406.15126","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:d386c180580a5897f72c23db1c3362b52f38feb93623e5829a79b273cde95882","observation_id":"fcd82b9e-de32-48a2-8c7e-f66f1cbf1563","resolution":{"observed_at":"2026-08-07T00:14:06.967525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05121","last_updated":"2024-10-24T07:26:36Z","snapshot_observed_at":"2026-07-06T17:27:03.686119Z","submitted_at":"2024-02-04T00:47:53Z","title":"Large Language Model for Table Processing: A Survey","version":3},"cited_work":{"arxiv_id":"2402.05121","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.05121","snapshot_observed_at":"2026-08-07T00:14:11.124548Z","title":"Large Language Model for Table Processing: A Survey","venue":"cs.AI","work_id":"2d9bb33c-992e-4e81-91fb-f758615aad64","year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.112279Z"},"links":{"cited_paper":"/paper/2402.05121","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:778c1054c6a799f2551e3bb11963123f5f3f401df040b1b7996d66e2d2672d20","observation_id":"c16a68d9-0780-45fb-876f-5879378d7d40","resolution":{"observed_at":"2026-08-07T00:14:11.249278Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.09140","last_updated":"2024-05-31T14:28:40Z","snapshot_observed_at":"2026-08-10T01:09:58.582440Z","submitted_at":"2022-04-19T21:55:18Z","title":"Multi-hop Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.09140","snapshot_observed_at":"2026-08-07T00:14:07.269681Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.269681Z"},"links":{"cited_paper":"/paper/2204.09140","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:bc67a3c140821bfe7f05b35916d635e6513def478ede20148225aad38cc999f9","observation_id":"3f88b3e2-6419-40bb-93d1-4a3cfda7e118","resolution":{"observed_at":"2026-08-07T00:14:07.269681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10150","last_updated":"2024-04-15T21:42:20Z","snapshot_observed_at":"2026-08-10T01:21:33.359222Z","submitted_at":"2024-04-15T21:42:20Z","title":"TabSQLify: Enhancing Reasoning Capabilities of LLMs Through Table Decomposition","version":1},"cited_work":{"arxiv_id":"2404.10150","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.10150","snapshot_observed_at":"2026-08-07T00:14:10.956785Z","title":"TabSQLify: Enhancing Reasoning Capabilities of LLMs Through Table Decomposition","venue":"cs.CL","work_id":"f726a201-ef27-4cc9-9c5d-af729d4e5685","year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.438053Z"},"links":{"cited_paper":"/paper/2404.10150","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:bec0381f0cdab0dfff90ce1c2b339285fe659def6021182a29ef79c64336ba8d","observation_id":"19b10a9b-6259-47d5-ac26-179e5fd0e4aa","resolution":{"observed_at":"2026-08-07T00:14:11.034685Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T00:14:07.558906Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.558906Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:4fce3a4cd4d2bb7ec27e8908a7f0066e02bb61b45b09fbad2ebe4314e6f3f327","observation_id":"2a746bde-c95a-4f84-9533-c286a50424b6","resolution":{"observed_at":"2026-08-07T00:14:07.558906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-07T00:14:07.667857Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.667857Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:04467fb08ba9a85cdd9e724c9943417449b50c45ffa810910cfa14f079a4f5ff","observation_id":"2e25baf2-0f38-4201-bad2-e033ecd2d313","resolution":{"observed_at":"2026-08-07T00:14:07.667857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09836","last_updated":"2023-11-16T12:05:23Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:05:23Z","title":"PELMS: Pre-training for Effective Low-Shot Multi-Document Summarization","version":1},"cited_work":{"arxiv_id":"2311.09836","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.09836","snapshot_observed_at":"2026-08-07T00:14:10.723447Z","title":"PELMS: Pre-training for Effective Low-Shot Multi-Document Summarization","venue":"cs.CL","work_id":"c02dc7d2-7d46-4b9f-aa89-157a92b474f9","year":2023},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.734524Z"},"links":{"cited_paper":"/paper/2311.09836","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:7d564baf7f7641ac373df57138e96b757c42e9641aa3cd9801b0336e40e1bb8a","observation_id":"96593aad-a964-4591-81dd-766059e73cbd","resolution":{"observed_at":"2026-08-07T00:14:10.818412Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13397","last_updated":"2024-06-19T09:38:59Z","snapshot_observed_at":"2026-08-10T07:58:29.506784Z","submitted_at":"2024-06-19T09:38:59Z","title":"MoreHopQA: More Than Multi-hop Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13397","snapshot_observed_at":"2026-08-07T00:14:07.861379Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:07.861379Z"},"links":{"cited_paper":"/paper/2406.13397","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:3304d4eadb50fcecb0299a88e7b3feba97cdfca471569b74848b6e33bb308bef","observation_id":"c830e442-5e36-4e51-bd5a-c27bd2bd3c40","resolution":{"observed_at":"2026-08-07T00:14:07.861379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16049","last_updated":"2024-03-23T21:21:44Z","snapshot_observed_at":"2026-08-04T21:08:49.454029Z","submitted_at":"2023-10-24T17:59:20Z","title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16049","snapshot_observed_at":"2026-08-07T00:14:08.038897Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.038897Z"},"links":{"cited_paper":"/paper/2310.16049","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:66408249ec2f9fa2ae8c403e50950553da8932fd0e5fb6509ee1ce6467b815d7","observation_id":"c54cfaf6-95a8-4a46-bd94-1599182cb0b4","resolution":{"observed_at":"2026-08-07T00:14:08.038897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T00:14:08.221512Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.221512Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:7f38e39d9eb1db4274512c2fb55b2bff54732f104c4e874dab244cbde26dfca6","observation_id":"8867b379-8bb7-4122-8cea-ebefc0f54f58","resolution":{"observed_at":"2026-08-07T00:14:08.221512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T00:14:08.351959Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.351959Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:fa978379eeb7370885537d5e29fdfd689250327ed5404c9628d4de557c965e65","observation_id":"96e63bb3-9ed9-49ad-ba4f-c3883d5718d4","resolution":{"observed_at":"2026-08-07T00:14:08.351959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.00573","last_updated":"2022-05-05T05:50:50Z","snapshot_observed_at":"2026-08-05T14:23:49.315145Z","submitted_at":"2021-08-02T00:33:27Z","title":"MuSiQue: Multihop Questions via Single-hop Question Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.00573","snapshot_observed_at":"2026-08-07T00:14:08.520930Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.520930Z"},"links":{"cited_paper":"/paper/2108.00573","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:762ad294d0206b132eb6e7b5217e358bdedca95155ca17859c2cea64a34c605f","observation_id":"22c514d7-0911-404c-81bf-824b5091ff20","resolution":{"observed_at":"2026-08-07T00:14:08.520930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"7741.12779","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:10.400347Z","title":null,"venue":null,"work_id":"58611a61-4596-4e56-9cdb-8df74c3ed30a","year":2007},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.678675Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:34b0cb9ec2b1eb44e0bad5202c3b74bce6818f4461a10e6119967d27c447b08f","observation_id":"d4ee0ef9-ada5-4c70-8b00-46dc4b3aa9df","resolution":{"observed_at":"2026-08-07T00:14:10.484753Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04398","last_updated":"2024-01-19T01:05:05Z","snapshot_observed_at":"2026-08-08T00:41:28.628631Z","submitted_at":"2024-01-09T07:46:26Z","title":"Chain-of-Table: Evolving Tables in the Reasoning Chain for Table Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04398","snapshot_observed_at":"2026-08-07T00:14:08.833582Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.833582Z"},"links":{"cited_paper":"/paper/2401.04398","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:872a59d8dab40a440252e623bffc4ce8efadbcc72a24d40453d43ae3c86c1d79","observation_id":"46b904af-a497-4a35-a455-7579b6c1c182","resolution":{"observed_at":"2026-08-07T00:14:08.833582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:08.980586Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:08.980586Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:cfc6731a4d3aa7f4075b6544e3cf9a784c0cb044554ab808ae48376275227b8b","observation_id":"c8d72a29-24e3-4eed-a32b-7b35e1327496","resolution":{"observed_at":"2026-08-07T00:14:08.980586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08499","last_updated":"2022-03-17T02:23:37Z","snapshot_observed_at":"2026-07-06T11:58:31.306092Z","submitted_at":"2021-10-16T07:22:24Z","title":"PRIMERA: Pyramid-based Masked Sentence Pre-training for Multi-document Summarization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.08499","snapshot_observed_at":"2026-08-07T00:14:09.155133Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.155133Z"},"links":{"cited_paper":"/paper/2110.08499","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:dcaedba2e2043b76c75d9272af32fa1e70d41076b52b39884095dce77185372c","observation_id":"97f2da7b-0a78-4a70-955b-96e8f16bfaba","resolution":{"observed_at":"2026-08-07T00:14:09.155133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06853","last_updated":"2024-10-08T15:55:37Z","snapshot_observed_at":"2026-07-06T17:14:57.853205Z","submitted_at":"2024-01-12T19:00:26Z","title":"Large Language Models Can Learn Temporal Reasoning","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06853","snapshot_observed_at":"2026-08-07T00:14:09.292502Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.292502Z"},"links":{"cited_paper":"/paper/2401.06853","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:360b94d810f83c7d36bb4a07d41daf5d29a00bc9a810fa1ec20d6487546b18f2","observation_id":"ad20fd23-15f5-4e36-8e37-0c492d017cc2","resolution":{"observed_at":"2026-08-07T00:14:09.292502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-07T00:14:09.408288Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.408288Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:9df96b57a317cd2db26cdbe15130b64df7f2aa25cc21e840e50ca1aeec4106bb","observation_id":"ffe69cda-a31e-4f18-94d3-3702c6ed9e42","resolution":{"observed_at":"2026-08-07T00:14:09.408288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:09.509757Z","title":"Cohen, Ruslan Salakhutdinov, and Christopher D","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.509757Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:1440cfd92a6a8a20fe0a0397e31ebd0f1fc86fedf070a1a4d64363edd8e7c3bd","observation_id":"13cf8528-eb54-4304-93a0-00bc37aae2d6","resolution":{"observed_at":"2026-08-07T00:14:09.509757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14116","last_updated":"2024-06-06T16:41:21Z","snapshot_observed_at":"2026-08-05T07:14:49.201751Z","submitted_at":"2024-02-21T20:30:45Z","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.14116","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.14116","snapshot_observed_at":"2026-08-07T00:14:10.025866Z","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","venue":"cs.CL","work_id":"2f082353-2bd7-4e0e-a62d-2375250c05c6","year":2024},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.575548Z"},"links":{"cited_paper":"/paper/2402.14116","citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:b13b8a85510214dd32e8a733d732a07c0845819b3e6f9d21abe949f6c0f88265","observation_id":"e5e4a797-0d26-4906-9fe7-574979813495","resolution":{"observed_at":"2026-08-07T00:14:10.137075Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:09.697027Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.697027Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:39b8f89e74c128c0f29705852046cb5a853949fc283c95ad6b7ebd98d8ecc121","observation_id":"4f4d466b-a270-4513-867d-32704d1d0b43","resolution":{"observed_at":"2026-08-07T00:14:09.697027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:14:09.828387Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:09.828387Z"},"links":{"citing_paper":"/paper/2506.14927"},"observation_digest":"sha256:0ba730816acb05e2ef2efe53f68ff8e36c0c320f6b4f03da3893d2a3746d4f77","observation_id":"16336597-edd6-4440-a1c2-430a51e1e722","resolution":{"observed_at":"2026-08-07T00:14:09.828387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.14927","last_updated":"2025-06-17T19:14:30Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T13:20:44.500546Z","submitted_at":"2025-06-17T19:14:30Z","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":26,"verified_exact":6,"verified_fuzzy":0},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 1 inbound Pith citation observation for arXiv:2506.14927."}