{"as_of":"2026-08-10T01:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bbe896c88720d6c717150ec35a45dcc947aa294038de8b9ff8c82ae4a938c7e4","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":43,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T00:53:29.674023Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":28,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-10T00:53:29.674023Z","title":"arXiv preprint arXiv:2308.11462","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.18062","last_updated":"2025-01-30T00:06:55Z","snapshot_observed_at":"2026-08-10T00:46:25.535610Z","submitted_at":"2025-01-30T00:06:55Z","title":"FinanceQA: A Benchmark for Evaluating Financial Analysis Capabilities of Large Language Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T00:53:29.674023Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2501.18062"},"observation_digest":"sha256:1e20bfac35175d4ae109bb4679cb7f751acb717e96a7052f11548b53b6a7e880","observation_id":"8a430616-9e27-4c79-a038-68fdfff5f118","resolution":{"observed_at":"2026-08-10T00:53:29.674023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-08T14:04:30.177659Z","title":"LegalBench: A collabora- tively built benchmark for measuring legal reasoning in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07022","last_updated":"2025-02-10T20:30:32Z","snapshot_observed_at":"2026-08-08T13:59:04.617930Z","submitted_at":"2025-02-10T20:30:32Z","title":"AIMS.au: A Dataset for the Analysis of Modern Slavery Countermeasures in Corporate Statements","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T14:04:30.177659Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2502.07022"},"observation_digest":"sha256:834f0a657cc6d35e6ccf38d227004ce8e33fbe798bb9bd1af2b0b7a0bafa2b05","observation_id":"a2a5231f-a917-4034-8073-9b3cf8d2e29f","resolution":{"observed_at":"2026-08-08T14:04:30.177659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-07T15:37:26.528544Z","title":"Ho, Christopher Ré, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14585","last_updated":"2025-09-04T11:31:24Z","snapshot_observed_at":"2026-08-07T21:21:35.164318Z","submitted_at":"2025-05-20T16:40:09Z","title":"Context Reasoner: Incentivizing Reasoning Capability for Contextualized Privacy and Safety Compliance via Reinforcement Learning","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T15:37:26.528544Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2505.14585"},"observation_digest":"sha256:fb58929137ac883d757a0b514558446fefa03a8b767b2f02018a74a955f709d2","observation_id":"067804ab-144d-4d0d-9762-7e1ba74c4a27","resolution":{"observed_at":"2026-08-07T15:37:26.528544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-07T14:21:43.049112Z","title":"Ho, Christopher Ré, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19259","last_updated":"2025-05-28T02:16:52Z","snapshot_observed_at":"2026-08-10T00:10:35.965934Z","submitted_at":"2025-05-25T18:28:12Z","title":"Towards Large Reasoning Models for Agriculture","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:21:43.049112Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2505.19259"},"observation_digest":"sha256:3553bdb7b31c901a83860ab43e6a2b2bd22344b4a36bb118978eda959a3ba74b","observation_id":"4457ddbc-cf62-45a7-bc22-44e191c38f93","resolution":{"observed_at":"2026-08-07T14:21:43.049112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-07T11:40:23.083854Z","title":"Tech- nical Report","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01671","last_updated":"2025-06-02T13:40:59Z","snapshot_observed_at":"2026-08-07T11:33:45.785028Z","submitted_at":"2025-06-02T13:40:59Z","title":"AIMSCheck: Leveraging LLMs for AI-Assisted Review of Modern Slavery Statements Across Jurisdictions","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:23.083854Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2506.01671"},"observation_digest":"sha256:9d5f244393eef533b4ad651a6b82d702dbc65997a79638c83b1cc0da642eef70","observation_id":"26293698-8d4e-4e0a-aad9-ac0f2333dea5","resolution":{"observed_at":"2026-08-07T11:40:23.083854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-07T06:02:10.295568Z","title":"Ho, Christopher Ré, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06084","last_updated":"2025-06-06T13:45:34Z","snapshot_observed_at":"2026-08-07T05:58:15.953060Z","submitted_at":"2025-06-06T13:45:34Z","title":"WisWheat: A Three-Tiered Vision-Language Dataset for Wheat Management","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T06:02:10.295568Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2506.06084"},"observation_digest":"sha256:abf68656aff8d2f3b0c559fd2477ed8cdad2fc82b271cd0f528a87ead33788c9","observation_id":"a0539aad-19f8-411e-b7d9-c78ec157b0f0","resolution":{"observed_at":"2026-08-07T06:02:10.295568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-06T22:56:32.246086Z","title":"LegalBench:ACollaborativelyBuiltBenchmarkforMeasuringLegalReasoning in Large Language Models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20274","last_updated":"2025-06-25T09:34:25Z","snapshot_observed_at":"2026-08-07T12:25:51.783364Z","submitted_at":"2025-06-25T09:34:25Z","title":"Enterprise Large Language Model Evaluation Benchmark","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:56:32.246086Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2506.20274"},"observation_digest":"sha256:cbf08d0f041fe940e763275b197292b0823febb2d2bc23d2c45856e045c37358","observation_id":"0c23ae7a-5ba5-4c5b-8fa2-fd50e0ddb26d","resolution":{"observed_at":"2026-08-06T22:56:32.246086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-06T18:35:40.873373Z","title":"arXiv preprint arXiv:2308.11462 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07893","last_updated":"2025-08-30T10:00:26Z","snapshot_observed_at":"2026-08-09T22:53:54.335664Z","submitted_at":"2025-07-10T16:22:41Z","title":"An Integrated Framework of Prompt Engineering and Multidimensional Knowledge Graphs for Legal Dispute Analysis","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T18:35:40.873373Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2507.07893"},"observation_digest":"sha256:18da0d2c3c1faef86f18e54528b411066fad9720238b356f0da035ced91b13e1","observation_id":"3eb6f863-281f-4072-b227-c83fca4980d6","resolution":{"observed_at":"2026-08-06T18:35:40.873373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T15:51:46.275259Z","title":"Ho, Christopher Ré, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.19365","last_updated":"2025-08-26T18:53:39Z","snapshot_observed_at":"2026-08-07T06:19:33.748455Z","submitted_at":"2025-08-26T18:53:39Z","title":"AI for Statutory Simplification: A Comprehensive State Legal Corpus and Labor Benchmark","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T15:51:46.275259Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2508.19365"},"observation_digest":"sha256:09a1c0ce0eab40a4ee0b1f4d1a81e61a12bbc370e65c63264a58c59d9e59005d","observation_id":"9ce16b5a-141e-4626-be52-158ad33feedf","resolution":{"observed_at":"2026-08-05T15:51:46.275259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T13:20:40.339812Z","title":"Ho, et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00761","last_updated":"2026-07-15T19:38:53Z","snapshot_observed_at":"2026-08-06T22:48:20.907522Z","submitted_at":"2025-08-31T09:23:26Z","title":"L-MARS: Legal Multi-Agent System with Agentic Search and Citation-Faithfulness Audit","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T13:20:40.339812Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2509.00761"},"observation_digest":"sha256:5fdb454f915a348f78a4d492a84d2c392b2832a0dd3e873baae7ede2daecd99c","observation_id":"0efc6177-c0c8-4f41-8fd8-c92df459e153","resolution":{"observed_at":"2026-08-05T13:20:40.339812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2510.26083","last_updated":"2026-04-08T13:55:00Z","snapshot_observed_at":"2026-07-06T22:34:28.484520Z","submitted_at":"2025-10-30T02:41:54Z","title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T03:05:06.069642Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2510.26083"},"observation_digest":"sha256:cc1b95050c84a67a8594ed9294351819bf87c73125b2755d83433e8de8957ad5","observation_id":"19fa69a8-2765-403f-8f13-90135d012582","resolution":{"observed_at":"2026-05-18T03:05:48.097934Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2511.05501","last_updated":"2026-04-28T16:06:15Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-09-30T21:36:23Z","title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T10:59:16.139525Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2511.05501"},"observation_digest":"sha256:5944965e4c491c73dbaa5827f4b98a0992546931d642822af777e8de87ea61f8","observation_id":"11d79624-30eb-429a-9918-0e82de232510","resolution":{"observed_at":"2026-05-18T11:01:17.029897Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2603.06610","last_updated":"2026-05-22T08:27:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-19T09:46:24Z","title":"CapTrack: Multifaceted Evaluation of Forgetting in LLM Post-Training","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T06:40:51.046965Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2603.06610"},"observation_digest":"sha256:180b9f14b4b7e085f664e7e31555f6fa9df95c23f7ca56d15de314c6de1d1547","observation_id":"2fb5a109-91ee-4c48-9816-733b21e74e4e","resolution":{"observed_at":"2026-05-25T06:45:25.582582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.10718","last_updated":"2026-04-12T16:28:51Z","snapshot_observed_at":"2026-08-06T20:40:21.653968Z","submitted_at":"2026-04-12T16:28:51Z","title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:34.768853Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.10718"},"observation_digest":"sha256:e39fdacaad262f68f551a298a0544b8039b665363c4a8f640a4e63809f8e8769","observation_id":"0770fa1b-4680-4353-8f8d-7a5cd87c9fbc","resolution":{"observed_at":"2026-05-11T09:36:03.821732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.18878","last_updated":"2026-04-20T22:00:02Z","snapshot_observed_at":"2026-07-06T23:05:39.724237Z","submitted_at":"2026-04-20T22:00:02Z","title":"LegalBench-BR: A Benchmark for Evaluating Large Language Models on Brazilian Legal Decision Classification","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T04:27:23.053818Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.18878"},"observation_digest":"sha256:1ca35c1be18c7ff633339a830e60b75d97c9af0bffc309b3f6c1855263b5c523","observation_id":"e0c78989-4ba3-4b36-a240-7692ee3934dc","resolution":{"observed_at":"2026-05-11T11:56:26.755806Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.19820","last_updated":"2026-04-19T07:09:42Z","snapshot_observed_at":"2026-07-06T23:06:21.774788Z","submitted_at":"2026-04-19T07:09:42Z","title":"KnowPilot: Your Knowledge-Driven Copilot for Domain Tasks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T06:15:44.360621Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.19820"},"observation_digest":"sha256:14a2ce05a3524dc311dca95ab0cda8048d265fe296c233e195fda76b0bbf7e5b","observation_id":"e4704eac-14b1-4ed5-9f66-7916bd338bd6","resolution":{"observed_at":"2026-05-10T06:16:20.746297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.19895","last_updated":"2026-04-21T18:17:08Z","snapshot_observed_at":"2026-07-06T23:06:26.596990Z","submitted_at":"2026-04-21T18:17:08Z","title":"Learning When Not to Decide: A Framework for Overcoming Factual Presumptuousness in AI Adjudication","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T02:08:24.770003Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.19895"},"observation_digest":"sha256:362f8c2d6b1c3981c1c35657336541f9833e0606750b3e38732ae5be500cb672","observation_id":"398924d6-1683-4640-a4fd-9614f707dddd","resolution":{"observed_at":"2026-05-10T02:11:57.255230Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.20726","last_updated":"2026-04-23T08:13:50Z","snapshot_observed_at":"2026-08-02T13:38:33.092933Z","submitted_at":"2026-04-22T16:12:36Z","title":"Exploiting LLM-as-a-Judge Disposition on Free Text Legal QA via Prompt Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T01:10:56.318664Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.20726"},"observation_digest":"sha256:7260a50556d711d2cf27948ff56902666ca7ef2a14667c8dda676c044edb5be8","observation_id":"a7818266-5c70-4bb6-b044-e164685aefa9","resolution":{"observed_at":"2026-05-11T13:41:10.502069Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.23511","last_updated":"2026-04-26T03:13:47Z","snapshot_observed_at":"2026-07-06T23:09:43.659053Z","submitted_at":"2026-04-26T03:13:47Z","title":"Breaking the Secret: Economic Interventions for Combating Collusion in Embodied Multi-Agent Systems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-08T06:13:17.633146Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.23511"},"observation_digest":"sha256:638a0ee430a9173d2a4846f1b8cd7b42f90bbc05abb5f40125fb04b7c8efde45","observation_id":"cf967cc5-c82b-4272-8813-f631e1ac7063","resolution":{"observed_at":"2026-05-11T21:16:16.378379Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.23730","last_updated":"2026-04-26T14:15:43Z","snapshot_observed_at":"2026-07-31T16:26:34.698889Z","submitted_at":"2026-04-26T14:15:43Z","title":"Expert Evaluation of LLM's Open-Ended Legal Reasoning on the Japanese Bar Exam Writing Task","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T05:59:54.438189Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.23730"},"observation_digest":"sha256:eee64b859883912cf282f4159360af8ecf007d02e4916e0d0276bc802fb618a9","observation_id":"1fd275bc-c7b0-44e8-8671-6e4c428bb6cd","resolution":{"observed_at":"2026-05-11T21:16:36.010805Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.24902","last_updated":"2026-04-27T18:34:08Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T18:34:08Z","title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-07T17:53:57.169962Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.24902"},"observation_digest":"sha256:0ac85300bcd26fcaa6693f026150f2f383c075d0b31ae51675005e7ff1a06f2d","observation_id":"ebcbdeb7-2d0d-4972-9bbf-2ed349ec7aad","resolution":{"observed_at":"2026-05-11T23:11:19.337354Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.07096","last_updated":"2026-06-04T16:41:18Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T01:24:06Z","title":"Query-efficient model evaluation using cached responses","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T23:28:47.530333Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.07096"},"observation_digest":"sha256:75a7f43d9c3c34079ca3df8a16d3162f928fad16e2e597bf1c32f31174887ab0","observation_id":"11fe66d3-1f22-46f5-8cfc-2863a2f81a9c","resolution":{"observed_at":"2026-06-30T23:35:07.682388Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.09611","last_updated":"2026-05-10T15:48:38Z","snapshot_observed_at":"2026-07-06T23:21:39.966502Z","submitted_at":"2026-05-10T15:48:38Z","title":"Byte-Exact Deduplication in Retrieval-Augmented Generation: A Three-Regime Empirical Analysis Across Public Benchmarks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T04:18:11.836537Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.09611"},"observation_digest":"sha256:0cfe78488c403a6d08817077ab67b2c056c8d94ff689a7d8d9b5a0a80634e50a","observation_id":"2d676823-a8f4-4cee-8652-9637cce9eb42","resolution":{"observed_at":"2026-05-12T06:26:24.208127Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.24454","last_updated":"2026-05-23T08:03:31Z","snapshot_observed_at":"2026-08-03T03:54:24.253471Z","submitted_at":"2026-05-23T08:03:31Z","title":"Decompose-and-Refine: Structured Legal Question Answering with Parametric Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T13:23:16.122654Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.24454"},"observation_digest":"sha256:5e64b07c5129533d7d3de139cd484c072fca8deecfded4fb27aade936341fe29","observation_id":"f34661f7-0ff0-484e-953f-b59557e57208","resolution":{"observed_at":"2026-06-30T13:24:39.984446Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.25474","last_updated":"2026-05-25T06:26:46Z","snapshot_observed_at":"2026-08-05T06:19:01.106812Z","submitted_at":"2026-05-25T06:26:46Z","title":"TypedCSIP: Typed Counterfactual Pretraining for Chinese Legislative Conflict Classification","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:06:28.247167Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.25474"},"observation_digest":"sha256:ba30a3352c99032e112d0645cc88ca5b47ebe055fbaa49fadfa5e4ec4f4ced6c","observation_id":"0a20cf91-1c6c-49fb-80ab-640f03d4acf5","resolution":{"observed_at":"2026-06-29T22:14:00.218650Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.29170","last_updated":"2026-05-27T23:12:20Z","snapshot_observed_at":"2026-07-06T23:38:37.178378Z","submitted_at":"2026-05-27T23:12:20Z","title":"UA-Legal-Bench: A Benchmark for Evaluating Large Language Models on Ukrainian Legal Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T12:15:12.570661Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.29170"},"observation_digest":"sha256:1ac795a99ea30bb195033bfdbd43b6a4691928945e0c627e279d22de9d7ede72","observation_id":"762cbd88-088c-4f89-a677-07649586ee0f","resolution":{"observed_at":"2026-06-29T12:23:24.527606Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.29738","last_updated":"2026-08-07T16:55:49Z","snapshot_observed_at":"2026-08-10T01:14:20.845792Z","submitted_at":"2026-05-28T10:31:37Z","title":"Multi-Legal-Bench: Evaluating LLMs on Legal Reasoning Across Jurisdictions, Languages, and Legal Traditions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T07:24:08.269519Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.29738"},"observation_digest":"sha256:827be83b38af04fcd77ea140c73325fd66e83c7b5643557a5d3b498959299fbb","observation_id":"89084092-98aa-40f3-ae4c-8eaade03f412","resolution":{"observed_at":"2026-06-29T08:33:15.935573Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.00898","last_updated":"2026-05-30T21:22:47Z","snapshot_observed_at":"2026-08-07T09:20:29.263336Z","submitted_at":"2026-05-30T21:22:47Z","title":"Citation Grounding: Detecting and Reducing LLM Citation Hallucinations via Legal Citation Graphs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T18:37:22.299236Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.00898"},"observation_digest":"sha256:6ec36919a2b6abb464d1cf8042e2ba8816c1ffc7728bf176fd70b67b76362c9d","observation_id":"0e17b22c-bbd3-4541-94e6-7a937abef3c6","resolution":{"observed_at":"2026-06-28T20:32:37.522670Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.08036","last_updated":"2026-06-06T07:56:40Z","snapshot_observed_at":"2026-08-06T03:27:44.094419Z","submitted_at":"2026-06-06T07:56:40Z","title":"GIScholarBench: Benchmarking LLM Overconfidence in GIS Research","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T19:17:26.825374Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.08036"},"observation_digest":"sha256:ac26ec052333183586b3d4eaf9dd965ebc26b23363e9e61a6e5d1b338b10d0fc","observation_id":"c02298e6-473d-41c6-a231-e1f3028cde94","resolution":{"observed_at":"2026-07-02T22:07:25.908133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.10457","last_updated":"2026-06-09T06:05:29Z","snapshot_observed_at":"2026-08-08T02:04:08.594731Z","submitted_at":"2026-06-09T06:05:29Z","title":"Trace2Policy: From Expert Behavior Traces to Self-Evolving Decision Agents","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T13:19:10.714343Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.10457"},"observation_digest":"sha256:d619d8f628fa70bdc13979cc6c7a6eab09122b86b8cc67e1653c5d2404c9516d","observation_id":"11bcab69-c3de-42e6-b065-a10de6e042fb","resolution":{"observed_at":"2026-07-03T05:27:39.663246Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.18021","last_updated":"2026-06-16T15:02:37Z","snapshot_observed_at":"2026-08-07T04:41:26.624398Z","submitted_at":"2026-06-16T15:02:37Z","title":"LegalHalluLens: Typed Hallucination Auditing and Calibrated Multi-Agent Debate for Trustworthy Legal AI","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-27T00:34:53.489694Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.18021"},"observation_digest":"sha256:66c03c2f573e08d984f89563befa9ff505799f081a588af22ba5ab428575fd6f","observation_id":"39b03420-626a-435f-bb7b-9451693948dd","resolution":{"observed_at":"2026-07-03T21:28:58.612291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.18158","last_updated":"2026-06-16T16:57:12Z","snapshot_observed_at":"2026-08-03T19:45:28.636780Z","submitted_at":"2026-06-16T16:57:12Z","title":"The Measurement Gap in the Automation of EU Law: Benchmarking Doctrinal Legal Reasoning under the EU AI Act","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-26T22:16:25.919667Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.18158"},"observation_digest":"sha256:4e0dc0dad4a42f0a9c1fc5fda88d2e9f1b5eea457c85408c2d11e907dba896c6","observation_id":"c454b269-996a-4f29-b290-fd070686884d","resolution":{"observed_at":"2026-07-03T23:29:02.495106Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.21121","last_updated":"2026-06-19T05:47:44Z","snapshot_observed_at":"2026-08-04T13:33:33.678027Z","submitted_at":"2026-06-19T05:47:44Z","title":"Answer Engineering: Local Trajectory Editing for Protocol-Constrained Decision Making in Large Language Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-06-26T14:13:13.678245Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.21121"},"observation_digest":"sha256:3617388f4838e6a0521f8cfd580dd9a9b9301489c689b972eed82cdab9e4325a","observation_id":"5d0442ac-aa2b-455d-935b-99fef5a8c367","resolution":{"observed_at":"2026-07-04T06:49:37.576858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.22778","last_updated":"2026-06-22T02:42:06Z","snapshot_observed_at":"2026-08-08T12:33:48.957919Z","submitted_at":"2026-06-22T02:42:06Z","title":"HAKARI-Bench: A Lightweight Benchmark for Comparing Retrieval Architectures and Efficiency Settings under Unified Conditions","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-06-26T07:22:34.547816Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.22778"},"observation_digest":"sha256:5d74d06fff53aa63b65eb2be8d2daf820cc9616c0af1a692f117cede53319397","observation_id":"9675f5fb-93ac-48c5-bd49-3d74a374a686","resolution":{"observed_at":"2026-07-04T11:59:51.263031Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.23716","last_updated":"2026-06-16T14:19:04Z","snapshot_observed_at":"2026-08-03T11:45:50.371507Z","submitted_at":"2026-06-16T14:19:04Z","title":"Legal Reasoning Is Not Lawyering: Rethinking Legal Benchmarks for Pro Se Access to Justice","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-26T22:26:54.239505Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.23716"},"observation_digest":"sha256:98d167921218ba7eb1f083e9a6986d30215c091e79fb6ccfdd3da1facd2cc212","observation_id":"77a56eeb-1020-40e3-b02d-1e0bc4c61849","resolution":{"observed_at":"2026-07-03T23:19:03.985515Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":"arXiv (Cornell University)","work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.26346","last_updated":"2026-06-24T19:38:21Z","snapshot_observed_at":"2026-08-08T08:43:50.725073Z","submitted_at":"2026-06-24T19:38:21Z","title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T01:34:10.103638Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.26346"},"observation_digest":"sha256:19b455c1892f405684c032be2fb0958cc6736ed39207e7e845406f0166724435","observation_id":"58c35bc4-9e48-487d-bae7-3e457539c130","resolution":{"observed_at":"2026-07-04T15:29:56.715847Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-14T10:38:03.549728Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10588","last_updated":"2026-07-12T05:57:22Z","snapshot_observed_at":"2026-08-09T00:27:00.303736Z","submitted_at":"2026-07-12T05:57:22Z","title":"Constraint-Aware Hierarchical Search for Regulation-Driven Fine-Grained Classification","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-07-14T10:38:03.549728Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.10588"},"observation_digest":"sha256:fa2fb1d5ae94eedcea1ac267f8238cd971c5378e8284fc90b14e4bcd7c8a51f4","observation_id":"b2379fd8-9d26-4202-822d-e816f773e61c","resolution":{"observed_at":"2026-07-14T10:38:03.549728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T19:01:02.854545Z","title":"and Ré, Christopher and Chilton, Adam and Narayana, Aditya and Chohlas-Wood, Alex and Peters, Austin and Waldon, Brandon and Rockmore, Daniel N","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17111","last_updated":"2026-07-19T07:41:46Z","snapshot_observed_at":"2026-08-03T13:30:27.639861Z","submitted_at":"2026-07-19T07:41:46Z","title":"BLAD: A Historically Contextualized, Multilingual Dataset of Bangladeshi Legal Acts (1799 to 2025)","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-01T19:01:02.854545Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.17111"},"observation_digest":"sha256:939f242a1943a8e3c3801c3e0e737fa3103a4c69907e816e8e30ecc894abf5e1","observation_id":"0f1c2e0f-2e20-40e7-897e-9605cfbc2a4f","resolution":{"observed_at":"2026-08-01T19:01:02.854545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T14:17:23.403369Z","title":"arXiv preprint arXiv:2308.11462 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18825","last_updated":"2026-07-21T08:01:17Z","snapshot_observed_at":"2026-08-08T11:57:56.292573Z","submitted_at":"2026-07-21T08:01:17Z","title":"AILQA: Evaluating AI-Driven Legal Question Answering Systems for the Indian Legal System","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-01T14:17:23.403369Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.18825"},"observation_digest":"sha256:818904b08f81c7f0ec27bfda6615d1a45890b2fbce7f1a652f6342a2c1f3981d","observation_id":"7971769f-9742-4175-9f19-32ee4b0d2009","resolution":{"observed_at":"2026-08-01T14:17:23.403369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T04:04:13.822723Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22954","last_updated":"2026-07-24T23:43:35Z","snapshot_observed_at":"2026-08-07T17:18:57.744300Z","submitted_at":"2026-07-24T23:43:35Z","title":"Toward Automated Detection of Documentation Inconsistencies in Electronic Health Records","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T04:04:13.822723Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.22954"},"observation_digest":"sha256:644aa13ae32dcf94ff624e5e240739eb1af2de0eb608b98d75e64ef4a98f0baa","observation_id":"0b3da14e-438a-45a5-b651-7b16132da10e","resolution":{"observed_at":"2026-08-01T04:04:13.822723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-30T23:41:20.109043Z","title":"Ho, Christopher Ré, Adam Chilton, Alex Chohlas-Wood, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23386","last_updated":"2026-07-25T22:40:02Z","snapshot_observed_at":"2026-08-08T16:52:19.365177Z","submitted_at":"2026-07-25T22:40:02Z","title":"Confidently Wrong: Exception Chain Collapse in Frontier LLM Rule Evaluation","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-07-30T23:41:20.109043Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.23386"},"observation_digest":"sha256:41203bea014432856e05ca15a61d15d3897915d66bec375b1e3ce7c86e0268a4","observation_id":"10731c04-647d-4086-9b51-aa77a094d1c9","resolution":{"observed_at":"2026-07-30T23:41:20.109043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T01:32:21.858955Z","title":"2023 , bdsk-url-1 =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.27783","last_updated":"2026-07-30T07:14:50Z","snapshot_observed_at":"2026-08-06T12:17:00.670693Z","submitted_at":"2026-07-30T07:14:50Z","title":"Reasoning Consensus: Structural Ensembling of LLM Reasoning via Weighted DAG Aggregation","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-01T01:32:21.858955Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.27783"},"observation_digest":"sha256:7660c1713cf4547a55877fdeaa383abe0e5953dcd597f1a3090fde57588d9376","observation_id":"1557386e-86b7-4762-b185-713397cfc408","resolution":{"observed_at":"2026-08-01T01:32:21.858955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-06T00:08:23.331668Z","title":"E.; Ré, C.; Chilton, A.; Narayana, A.; Chohlas-Wood, A.; Peters, A.; Waldon, B.; Rockmore, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01522","last_updated":"2026-08-02T22:19:01Z","snapshot_observed_at":"2026-08-09T08:15:52.624300Z","submitted_at":"2026-08-02T22:19:01Z","title":"Question Begets Question: Self-Evolving Curriculum for Reinforcement Fine-Tuning on Competition Mathematics","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T00:08:23.331668Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2608.01522"},"observation_digest":"sha256:ccff411277061eda0ab3d12bc68b0c7755808561eb0298887bffb191497cbc8b","observation_id":"082e6d06-398f-4600-a127-18d04123a176","resolution":{"observed_at":"2026-08-06T00:08:23.331668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.11462/citation-record","integrity":"/paper/2308.11462/integrity","json":"/paper/2308.11462/citation-record.json","paper":"/paper/2308.11462"},"outbound":[],"paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T20:35:47.668090Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 43 inbound Pith citation observations for arXiv:2308.11462."}