{"as_of":"2026-08-15T23:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f8f8a520df9ff158323475d0b7582c244ca79a4c9f61757b483d56a266d00f9c","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T17:26:38.394751Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.10963/citation-record","integrity":"/paper/2509.10963/integrity","json":"/paper/2509.10963/citation-record.json","paper":"/paper/2509.10963"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.274545Z","title":"A., and Sheikh, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.274545Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:ce5bf0e76c18af94171d555cf40a2c8c3c495c065d26dc5502a1b6346ab35f58","observation_id":"c83ea906-353f-44e0-b624-c9fe36a54cbd","resolution":{"observed_at":"2026-08-04T17:26:38.274545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-04T17:26:38.281877Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.281877Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:0b9727100380c3213968f2749d4b0aad551f944b480ceda322ebd199bf2b5e84","observation_id":"6f44d130-7c88-43dc-91fd-b7eabc4621cb","resolution":{"observed_at":"2026-08-04T17:26:38.281877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08784","last_updated":"2026-07-24T02:23:43Z","snapshot_observed_at":"2026-08-15T21:44:09.339269Z","submitted_at":"2025-05-13T17:58:16Z","title":"PCS-UQ: Uncertainty Quantification via the Predictability-Computability-Stability Framework","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.08784","snapshot_observed_at":"2026-08-04T17:26:38.286778Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.286778Z"},"links":{"cited_paper":"/paper/2505.08784","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:7f573ac5e03399723de454d4e7906d3b299ee3e1b937d3b8411dcf87941de69b","observation_id":"92eb883e-14b0-40b7-ba1f-cf10bb00367e","resolution":{"observed_at":"2026-08-04T17:26:38.286778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.291673Z","title":null,"venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.291673Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:34fa489b2bdb71b363eca0b9a25959fcc8451416e88e51b34c2b7f9c5734db02","observation_id":"262ac843-df59-4488-b0a3-78a6a830d550","resolution":{"observed_at":"2026-08-04T17:26:38.291673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.295941Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.295941Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:f987b0a924144af9bcc41bc55d61af24a538b408e0cdec21c8c4c3d74a01866e","observation_id":"5ff41350-95a1-4dd6-a4b6-e4441380da6f","resolution":{"observed_at":"2026-08-04T17:26:38.295941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08939","last_updated":"2024-05-28T04:32:09Z","snapshot_observed_at":"2026-08-13T04:19:21.806858Z","submitted_at":"2024-02-14T04:50:18Z","title":"Premise Order Matters in Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08939","snapshot_observed_at":"2026-08-04T17:26:38.300315Z","title":"A., Wang, X., and Zhou, D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.300315Z"},"links":{"cited_paper":"/paper/2402.08939","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:64b695d7cf7afd41d8569fc7eeb99707066a42198d2eb812d8bc7087f5bda747","observation_id":"ee658f95-f988-4f5e-81c7-455f8ada20ba","resolution":{"observed_at":"2026-08-04T17:26:38.300315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.305563Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.305563Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:50342fcae6cf9486b1292869bbe4942310e55154b08a36be3e8d64857fa8cd47","observation_id":"b4326488-5829-44d2-8156-94dab54f8307","resolution":{"observed_at":"2026-08-04T17:26:38.305563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12066","last_updated":"2024-06-19T03:59:41Z","snapshot_observed_at":"2026-08-12T23:40:38.398109Z","submitted_at":"2024-06-17T20:09:24Z","title":"Language Models are Surprisingly Fragile to Drug Names in Biomedical Benchmarks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12066","snapshot_observed_at":"2026-08-04T17:26:38.309880Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.309880Z"},"links":{"cited_paper":"/paper/2406.12066","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:69ff144d41563d0f97366789f997437c8ad8d9223b2f0873b45b04aef714231d","observation_id":"17d0b253-d3ad-4e5a-b560-e53b2b545bf9","resolution":{"observed_at":"2026-08-04T17:26:38.309880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-04T17:26:38.314279Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.314279Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:c42d8393e24058ee70bab68edaf3790f3d5f78474d78337f394899f5187869eb","observation_id":"60088fa0-5e18-4d49-aeaf-b07d58503228","resolution":{"observed_at":"2026-08-04T17:26:38.314279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.319011Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.319011Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:f6618d75c3820e1c4f0368b11ba10653f4e0bcadbd41407be2082a5a7c131acf","observation_id":"38076c9e-4f16-45cd-a402-be49e2cfe0ee","resolution":{"observed_at":"2026-08-04T17:26:38.319011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08913","last_updated":"2023-09-16T07:36:07Z","snapshot_observed_at":"2026-08-13T10:12:26.649215Z","submitted_at":"2023-09-16T07:36:07Z","title":"A Statistical Turing Test for Generative Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08913","snapshot_observed_at":"2026-08-04T17:26:38.323143Z","title":"E., and Yang, W","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.323143Z"},"links":{"cited_paper":"/paper/2309.08913","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:0d5400b8876bd35d8215df01b6069efc6652066011cf7dca8bbd2e31157f4b3d","observation_id":"f019bd5b-25b9-4d3a-af87-9a9adb3d788a","resolution":{"observed_at":"2026-08-04T17:26:38.323143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08825","last_updated":"2024-03-08T02:49:12Z","snapshot_observed_at":"2026-08-13T05:50:09.391590Z","submitted_at":"2023-10-13T02:41:55Z","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08825","snapshot_observed_at":"2026-08-04T17:26:38.328095Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.328095Z"},"links":{"cited_paper":"/paper/2310.08825","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:8c200f862a350cce196dd8105ba670682ba3671eb296a02157c4e5680fa67ebb","observation_id":"a54931fd-ea48-473d-a5de-b2c9e84bb2d0","resolution":{"observed_at":"2026-08-04T17:26:38.328095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.332606Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.332606Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:49f6cc4e55e90fdf735a7aa1b439c3c432e5b8211ec33be91a7617556762164f","observation_id":"4e6bcd10-9aa7-49db-b159-50e9b757ee75","resolution":{"observed_at":"2026-08-04T17:26:38.332606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.337094Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.337094Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:f46ff38944862f0b270e283064110bc99c9d44a4d1b3339e37eb55b8e3f7ad92","observation_id":"0936f829-08fc-4804-b0d8-bb73ab1b9fba","resolution":{"observed_at":"2026-08-04T17:26:38.337094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.341371Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.341371Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:46d3698bb6902dc7bf5f374bd08f08671ce45664c04183ad7ace812060005939","observation_id":"7f977a3c-2a57-46c3-823a-b8f29db191db","resolution":{"observed_at":"2026-08-04T17:26:38.341371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00640","last_updated":"2024-11-01T14:57:16Z","snapshot_observed_at":"2026-08-12T22:09:41.196978Z","submitted_at":"2024-11-01T14:57:16Z","title":"Adding Error Bars to Evals: A Statistical Approach to Language Model Evaluations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00640","snapshot_observed_at":"2026-08-04T17:26:38.345872Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.345872Z"},"links":{"cited_paper":"/paper/2411.00640","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:96775583dd73414307b185b052add1e213446474543d0ae7c2b5a3914bf13538","observation_id":"a649924f-008f-4c42-9079-f22fba09622d","resolution":{"observed_at":"2026-08-04T17:26:38.345872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06573","last_updated":"2024-09-01T19:38:02Z","snapshot_observed_at":"2026-08-12T23:51:12.193406Z","submitted_at":"2024-06-03T18:15:56Z","title":"MedFuzz: Exploring the Robustness of Large Language Models in Medical Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06573","snapshot_observed_at":"2026-08-04T17:26:38.350247Z","title":"O., Matton, K., Helm, H., Zhang, S., Bajwa, J., Priebe, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.350247Z"},"links":{"cited_paper":"/paper/2406.06573","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:50a462116ac6344b5b6aef8373ae3feaf5fbca8ab320789de989f1c332d58c25","observation_id":"d2966f35-8df5-4396-8403-1ab3f0feb77c","resolution":{"observed_at":"2026-08-04T17:26:38.350247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16452","last_updated":"2023-11-28T03:16:12Z","snapshot_observed_at":"2026-08-13T05:15:47.720293Z","submitted_at":"2023-11-28T03:16:12Z","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16452","snapshot_observed_at":"2026-08-04T17:26:38.354877Z","title":"T., Zhang, S., Carignan, D., Edgar, R., Fusi, N., King, N., Larson, J., Li, Y., Liu, W., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.354877Z"},"links":{"cited_paper":"/paper/2311.16452","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:20e9bf136ffa3fbc1ee84f73467334ef569e203ac5329de225e8e940f40964d1","observation_id":"797b5eb6-1460-48b0-b778-10461e3117c3","resolution":{"observed_at":"2026-08-04T17:26:38.354877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.359620Z","title":"M., Terano, H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.359620Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:3eac22f4074e542feac2ebe64ad8c98cdee1de8620fad5d5067870eb420e0c5f","observation_id":"0bba18bf-9cd8-46cb-8a49-697560c21547","resolution":{"observed_at":"2026-08-04T17:26:38.359620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.363860Z","title":"J., Ryan, P","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.363860Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:3d3c043eaf3d91fc66e604d038e60e87cb5ba9877a2b5b52f4a7e0fa22084fa5","observation_id":"b8dbb191-40e5-4b80-948f-30f759369290","resolution":{"observed_at":"2026-08-04T17:26:38.363860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16789","last_updated":"2024-03-09T22:26:06Z","snapshot_observed_at":"2026-08-08T18:07:29.632928Z","submitted_at":"2023-10-25T17:21:23Z","title":"Detecting Pretraining Data from Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16789","snapshot_observed_at":"2026-08-04T17:26:38.368248Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.368248Z"},"links":{"cited_paper":"/paper/2310.16789","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:b14d4a4ffb3162f0c336005b37f5618adbeb6c6d012dea8859bf09c41200faca","observation_id":"02110994-59fe-4827-af19-80ee369c0289","resolution":{"observed_at":"2026-08-04T17:26:38.368248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.372460Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.372460Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:65b3bf3ce9ed305323e51466c18472de0949447d84ff0af71f5e1c4b9e3d3051","observation_id":"46b0ba09-038a-44ab-81fe-2b23fdf69ec1","resolution":{"observed_at":"2026-08-04T17:26:38.372460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.09136","last_updated":"2023-03-16T08:01:22Z","snapshot_observed_at":"2026-08-15T14:08:59.323829Z","submitted_at":"2023-03-16T08:01:22Z","title":"A Short Survey of Viewing Large Language Models in Legal Aspect","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.09136","snapshot_observed_at":"2026-08-04T17:26:38.376545Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.376545Z"},"links":{"cited_paper":"/paper/2303.09136","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:399c5e95beb03127acc8b0c3d6d1be368b026d35817f8c90921c3395c3c21b64","observation_id":"e96bdd8e-e364-4eb7-95ea-4207c7361227","resolution":{"observed_at":"2026-08-04T17:26:38.376545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-04T17:26:38.381255Z","title":"M., Hauth, A., Millican, K., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.381255Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:5d20eb25852ca14ecffbb697cc8911af8fa898d8847f67137c1a886510b511db","observation_id":"5e31537c-3a1f-4ce6-902d-dcf2b6ab02da","resolution":{"observed_at":"2026-08-04T17:26:38.381255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07827","last_updated":"2024-02-12T17:34:13Z","snapshot_observed_at":"2026-08-15T16:46:04.358915Z","submitted_at":"2024-02-12T17:34:13Z","title":"Aya Model: An Instruction Finetuned Open-Access Multilingual Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07827","snapshot_observed_at":"2026-08-04T17:26:38.385609Z","title":"J., Ting, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.385609Z"},"links":{"cited_paper":"/paper/2402.07827","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:98eec50dff472e3966ce8cfa3914affa4a76da8d2cda28a0df297a76a3e27a69","observation_id":"3e159ec2-9d47-49dd-9ef2-f4cfb13fcec1","resolution":{"observed_at":"2026-08-04T17:26:38.385609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.390712Z","title":"and Barter, R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.390712Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:9e5f5ca007397e456db29aa5c34ef4d37add08b919cdb63ac92da48a9dd37be8","observation_id":"0026d3b8-97b2-4d65-801b-3de412dd33f2","resolution":{"observed_at":"2026-08-04T17:26:38.390712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07339","last_updated":"2024-08-09T06:16:55Z","snapshot_observed_at":"2026-08-13T04:43:10.179388Z","submitted_at":"2024-01-14T18:12:03Z","title":"CodeAgent: Enhancing Code Generation with Tool-Integrated Agent Systems for Real-World Repo-level Coding Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07339","snapshot_observed_at":"2026-08-04T17:26:38.394751Z","title":"rX k=1 X (j) k −X ′ k r −E rX k=1 X (j) k −X ′ k r ! < t # ≥1−2e − rt2 2 =⇒P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.394751Z"},"links":{"cited_paper":"/paper/2401.07339","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:1b7857480236a3163054c22d905ec27cbcb582a6bc9a37080f834d4f7e98dea6","observation_id":"227be055-c831-4a01-808e-66ccb3da17de","resolution":{"observed_at":"2026-08-04T17:26:38.394751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","latest_version":1,"primary_category":"math.ST","snapshot_observed_at":"2026-08-11T09:59:53.099472Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":27,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2509.10963."}