{"as_of":"2026-08-10T15:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b8ce7bf0850ab0e3a2eb5c8262785c0f87918bd6421f8b4272e8886de81a307c","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:51:32.423009Z","state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00064/citation-record","integrity":"/paper/2506.00064/integrity","json":"/paper/2506.00064/citation-record.json","paper":"/paper/2506.00064"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.590971Z","title":null,"venue":null,"work_id":"44472143-a5ec-49df-bbb0-aa87396c7417","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.431122Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:38bf5fd2c8076588a9b27c7a09d64e5628a5558de6f74ecdce4fa37d7cdc1411","observation_id":"cca2bcf8-2d83-4d32-9815-1be91e2442b6","resolution":{"observed_at":"2026-08-07T12:51:34.700354Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.295402Z","title":"primary-category","venue":null,"work_id":"2289c282-542a-4074-bf41-89d55eb40ef5","year":2006},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.491616Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:9bc470701eed25a9a4008ca2d2d29c07e75bdc4416f1d5f16fe6e2c130fe9f1c","observation_id":"b71a70cc-1971-4efe-b594-980cbe935a5f","resolution":{"observed_at":"2026-08-07T12:51:34.430233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"7582.3777","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:32.720039Z","title":"How long does it take to drive from the university to the beach?","venue":null,"work_id":"139dd36c-43f5-4f7a-bb23-db4513ed6b7f","year":2017},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.423009Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:bd1438dff38d3373509f368d50d1cb95935632fa3b239c703c0bf2ed079e0488","observation_id":"39511c44-53c8-4aa3-b1e6-fedb872d6332","resolution":{"observed_at":"2026-08-07T12:51:32.813447Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.829020Z","title":"Please ensure that the categorization is correct and not ambiguous","venue":null,"work_id":"397b3cdd-73c5-4014-862a-fe3d721e8b13","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.583967Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:32c6bcb0e6fb24bdf76f2d0bc8882791ebc0d3af843f7fbaaaa128d5862295d5","observation_id":"943478cc-e6db-48eb-ac9e-646001313517","resolution":{"observed_at":"2026-08-07T12:51:34.995301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.066583Z","title":null,"venue":null,"work_id":"21b85744-eaee-4f38-993e-4c7dba12e28d","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.678732Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:297975ef0bf2304eb82e1023267ed558a466d33a000189396348059b18a21338","observation_id":"7a63a5ca-9ff1-4132-90d7-65583dbac4a5","resolution":{"observed_at":"2026-08-07T12:51:34.138052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.861820Z","title":"primary-category","venue":null,"work_id":"9fb0e94b-5b35-49ef-a71c-8c7cb952da41","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.730040Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:0423d6d1325b6c24cb7e1cb9befaff3faee1d1a61e6b05d34aef5f67f8a5487f","observation_id":"25bded7d-f0b8-46f8-b04d-486ab57e748c","resolution":{"observed_at":"2026-08-07T12:51:33.986171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.805703Z","title":null,"venue":null,"work_id":"65bdc23b-b8cb-4689-9d01-748ee1276f07","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.778209Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:b667e83637f04cbc779a77b8c4084af0ac5569533bdf35f26ee6a9be87938152","observation_id":"1a5f419e-3793-4026-af34-9ae93102be26","resolution":{"observed_at":"2026-08-07T12:51:33.848324Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.690972Z","title":"Textual context and subject specification","venue":null,"work_id":"92ebb24b-4f66-4f20-8352-3af96ebba535","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.870624Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:d54fceaaf435effb4d953b79f4b0d7c2145c2e3b271ef09ad427f2ca7bacc864","observation_id":"af8b8a3d-71ee-495d-8a29-f7b751212fdb","resolution":{"observed_at":"2026-08-07T12:51:33.736624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.573706Z","title":"primary-category","venue":null,"work_id":"fddbfe9c-c0e1-44af-8ce4-ef30509da9b8","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.936392Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:d6512f2ff61a29206994a1db49edc6b8a3a6c4ebf974193998cc49bf848b1cbf","observation_id":"be8c2661-0265-4886-bc12-eb28d06142bd","resolution":{"observed_at":"2026-08-07T12:51:33.625222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.430572Z","title":null,"venue":null,"work_id":"12c58a28-49f8-4d30-a61e-b12498be22ac","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.986989Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:fd335f88b85bb4e76eb35e0879cea0934d38f3a1a981c9b1388162ea5774b796","observation_id":"cb79c9a2-471d-45b0-82ac-8630a27a4e7a","resolution":{"observed_at":"2026-08-07T12:51:33.502194Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.272222Z","title":"primary-category","venue":null,"work_id":"e8dc7e19-bf38-4c3e-80bb-6e26156d7b68","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.078046Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:8219cd4e2a48a4635de461890e7f25c69c043f4b272f519e77e6815033742b40","observation_id":"cdb70ed6-3a38-4bdc-a165-88eb4604d332","resolution":{"observed_at":"2026-08-07T12:51:33.357836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.140351Z","title":null,"venue":null,"work_id":"25a00210-96ee-4743-885c-4cf8a77c69c1","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.165554Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:696952e5791a92a9432e763b728eed4b22d600f1298b94686a2b38a301b29719","observation_id":"9080c926-6ff6-4e9d-a2d2-ead3c7242945","resolution":{"observed_at":"2026-08-07T12:51:33.199731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.049176Z","title":null,"venue":null,"work_id":"97c0e377-a2d2-4d57-84d9-5f6e0919b743","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.233456Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:a5b8cf124709f4e15223e85cfa081ca84443760f78a35029230d7d5da2646650","observation_id":"6a80cdb7-7ab5-4335-822f-2ae8e5aa2e32","resolution":{"observed_at":"2026-08-07T12:51:33.083109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:32.918080Z","title":null,"venue":null,"work_id":"8fbd80bc-de9c-4364-ad07-4ebbdf1b36df","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.342907Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:80d5faa5546d2dd3cb69b0f73276337bbf536048205004fb29890079044c2fe2","observation_id":"2987be69-2c7a-464a-a827-a700c925c63e","resolution":{"observed_at":"2026-08-07T12:51:33.000357Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19260","last_updated":"2025-01-02T18:46:05Z","snapshot_observed_at":"2026-08-10T05:28:04.707921Z","submitted_at":"2024-12-26T15:54:10Z","title":"MEDEC: A Benchmark for Medical Error Detection and Correction in Clinical Notes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19260","snapshot_observed_at":"2026-08-07T12:51:31.281190Z","title":"whear”, “histori","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.281190Z"},"links":{"cited_paper":"/paper/2412.19260","citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:da49c9527dd73ca1d7f892dcd329abb8bc063150ac5fd6e013774b1805ca87e7","observation_id":"472a067e-06df-4aa6-b763-7a6f01bde976","resolution":{"observed_at":"2026-08-07T12:51:31.281190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 0 inbound Pith citation observations for arXiv:2506.00064."}