{"as_of":"2026-08-11T03:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:51f0317d04a4995f5d16091633c136f20d6eca0b0a715d7810f00681b0fda053","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:08:46.326666Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:34:58.344234Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-14T00:18:29.517494Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-07T14:34:58.344234Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18434","last_updated":"2025-05-24T00:02:48Z","snapshot_observed_at":"2026-08-10T11:46:31.778304Z","submitted_at":"2025-05-24T00:02:48Z","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:34:58.344234Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2505.18434"},"observation_digest":"sha256:8149bc0a878aa9b7354cc887c18abf6a7a7771849aa91622aecc09ef66e1447f","observation_id":"6a51e034-03ea-4726-a7a1-89714e6234a0","resolution":{"observed_at":"2026-08-07T14:34:58.344234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-07T13:01:09.271049Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22946","last_updated":"2025-05-28T23:58:37Z","snapshot_observed_at":"2026-08-09T17:04:53.831723Z","submitted_at":"2025-05-28T23:58:37Z","title":"NegVQA: Can Vision Language Models Understand Negation?","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T13:01:09.271049Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2505.22946"},"observation_digest":"sha256:729db4c0d5ef6ab8c0e9ec1a62e9002bac7177f22844a80ec0f444991915fded","observation_id":"93ca8fa6-82d7-43c1-853c-34dbd8976c31","resolution":{"observed_at":"2026-08-07T13:01:09.271049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-04T20:28:21.003431Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09732","last_updated":"2025-09-10T13:08:03Z","snapshot_observed_at":"2026-08-09T18:45:49.898879Z","submitted_at":"2025-09-10T13:08:03Z","title":"Decomposing Visual Classification: Assessing Tree-Based Reasoning in VLMs","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T20:28:21.003431Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2509.09732"},"observation_digest":"sha256:1b1445fa652a17a58a6064a19edd89d71554396c54a4bd1a7a4ba3a9ce47e3a6","observation_id":"af5a841d-b63e-4378-9185-5f06f40dddbf","resolution":{"observed_at":"2026-08-04T20:28:21.003431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2604.18942","last_updated":"2026-04-21T00:32:18Z","snapshot_observed_at":"2026-07-06T23:05:39.724237Z","submitted_at":"2026-04-21T00:32:18Z","title":"Disparities In Negation Understanding Across Languages In Vision-Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T03:31:50.234201Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2604.18942"},"observation_digest":"sha256:74a2c44eba0c605b0bb16c3c28a4b80b3ce9d1e143f6d011cf05dca383f94092","observation_id":"65823167-c3b5-467e-9557-9740e1845312","resolution":{"observed_at":"2026-05-11T12:31:03.990536Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2604.22773","last_updated":"2026-03-31T03:05:26Z","snapshot_observed_at":"2026-07-06T23:09:05.682387Z","submitted_at":"2026-03-31T03:05:26Z","title":"Trace Mutation in Human-LLM Dialogue: The Transcript as Forensic and Mitigation Surface","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-14T00:18:14.010081Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2604.22773"},"observation_digest":"sha256:01efad3d499c32f557bcc9d450c827c9bc2f595aa980cefd4ec5a84ec5611387","observation_id":"8f28e7c4-ea6d-489f-935b-6c6cc2a6296d","resolution":{"observed_at":"2026-05-14T00:18:29.524884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2605.05810","last_updated":"2026-05-07T07:46:17Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T07:46:17Z","title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T14:49:53.357083Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2605.05810"},"observation_digest":"sha256:2574fa50e407e3a35c9b1ef472ef8b49cd6bfe879f0a3b81b129e61d5d49c290","observation_id":"2471455d-3f34-45b8-8239-b49ecff50ba2","resolution":{"observed_at":"2026-05-11T18:41:09.727332Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2605.06815","last_updated":"2026-05-07T18:16:49Z","snapshot_observed_at":"2026-08-04T01:42:42.718743Z","submitted_at":"2026-05-07T18:16:49Z","title":"Uneven Evolution of Cognition Across Generations of Generative AI Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-11T01:00:59.086900Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2605.06815"},"observation_digest":"sha256:ad655b38fac3b1113075d6330ac22628500d72e0abb8e93f7064e6f494e5070b","observation_id":"10e8c895-0807-4b6d-a917-130295408816","resolution":{"observed_at":"2026-05-11T04:50:58.108557Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-01T17:16:49.314099Z","title":"Available: https://arxiv.org/abs/2501.09425","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17712","last_updated":"2026-07-20T09:08:51Z","snapshot_observed_at":"2026-08-03T12:43:11.710062Z","submitted_at":"2026-07-20T09:08:51Z","title":"Learning to Detect Cross-Modal Negation: An Analysis of Latent Representations and an Attention-Based Solution","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T17:16:49.314099Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2607.17712"},"observation_digest":"sha256:1ca284e0f3619f504366af6b25588a293fb6628d0d23d1e7dccb3f45f2e5b77f","observation_id":"64c36674-3a95-4912-9f7c-3014d8fc0d3a","resolution":{"observed_at":"2026-08-01T17:16:49.314099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.09425/citation-record","integrity":"/paper/2501.09425/integrity","json":"/paper/2501.09425/citation-record.json","paper":"/paper/2501.09425"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.563891Z","title":"Aug- mented reality meets computer vision: Efficient data gen- eration for urban driving scenes","venue":null,"work_id":"bbd8caea-a74a-4a2b-8162-c4ed076d11d6","year":2018},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.008226Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:3d758322e273871cf55d0134286bf6ae0f4290f27d170c8bad64d4c982e494f2","observation_id":"876019cf-4e11-484d-93b7-b82286f09e37","resolution":{"observed_at":"2026-08-10T20:08:47.570497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.541877Z","title":"Effective conditioned and composed im- age retrieval combining clip-based features","venue":null,"work_id":"c1e8940f-4bde-4794-928a-0464d362e5e1","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.014749Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:e33c827e991ca7e960a6db4caa6648421f4cffeaa443370e00645af97f036cce","observation_id":"76626a12-b8c4-4540-861e-c24ddca9336b","resolution":{"observed_at":"2026-08-10T20:08:47.548229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.511347Z","title":"FitCLIP: Refining large- scale pretrained image-text models for zero-shot video un- derstanding tasks","venue":null,"work_id":"5dc3b93e-f045-4352-af50-c1e86fe03be9","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.020123Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:860d8c826f56af1a319cbffca94d08ef15b5e57231f0b5d540a98bf9876d918a","observation_id":"5d08414e-b6b6-492a-ac86-dbac81d7a1fe","resolution":{"observed_at":"2026-08-10T20:08:47.517244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.031905Z","title":"Conceptual 12M: Pushing web-scale image-text pre-training to recognize long-tail visual concepts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.031905Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:b3235bf29b730bba120fbb817b3ac68e98081ab08dd1e970debad589902dde15","observation_id":"adc40a9b-4ad3-44b7-96a0-b98ec2b7ff5d","resolution":{"observed_at":"2026-08-10T20:08:46.031905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.447175Z","title":"Learning semantic segmentation from synthetic data: A geo- metrically guided input-output adaptation approach","venue":null,"work_id":"a345b268-77a8-4cff-ab65-46842644f4d9","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.037424Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:8874ebbfb6faf548ab3b137250dfdb2f9f38d5a4fdd909b6c32fc55cc9454058","observation_id":"19db9dd0-2961-4ed7-ad64-5bae5646be84","resolution":{"observed_at":"2026-08-10T20:08:47.453953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-10T20:08:46.042836Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.042836Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:15a0a0c85327708e90c672349dea7b71f841f6bcda8fdc8142f4175a518b488f","observation_id":"3e0cc5c6-38ab-4ba6-855b-8a80c4030f3d","resolution":{"observed_at":"2026-08-10T20:08:46.042836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.421837Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"a7de39c3-229b-49d1-9e38-d9af2048a2c8","year":2010},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.048534Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:458df60a1a2b49f9f2e3b9e67df1705fe790f23ed3c216d926c5105048b90a22","observation_id":"97d27b26-b7c2-45d5-ab5c-dbe351a3529a","resolution":{"observed_at":"2026-08-10T20:08:47.429941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14402","last_updated":"2024-11-21T18:31:25Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:31:25Z","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14402","snapshot_observed_at":"2026-08-10T20:08:46.054919Z","title":"Mul- timodal autoregressive pre-training of large vision encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.054919Z"},"links":{"cited_paper":"/paper/2411.14402","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:7c77bc4a60401558073fb55135eef20fe783dff6b6c7e5fc16486690af0a66b3","observation_id":"11613278-2c26-4b96-9ae0-c64f4c515b2a","resolution":{"observed_at":"2026-08-10T20:08:46.054919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.395763Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":"d5915288-4623-4e5e-ab48-f707c03d31d4","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.061090Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f9e1e7e24ca11635ffe0cfe1d80ee2c0bb96e70cf96bc8c52f16a45a229980b3","observation_id":"43de77e5-ed52-44ba-8ed4-68d1bedef7ab","resolution":{"observed_at":"2026-08-10T20:08:47.402544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.373394Z","title":"This is not a dataset: A large negation benchmark to challenge large language mod- els","venue":null,"work_id":"3ccc5764-ddd9-4ab6-bb4e-6dfcd49a9c02","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.076557Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:a9bbd3075adf9fb7618c5ba604ffa58ef450d636a7b16dffe4eae5dd7546bb5c","observation_id":"e1117c67-0df2-4e42-a899-fb9d4b5c7bcd","resolution":{"observed_at":"2026-08-10T20:08:47.382504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.349563Z","title":"Shortcut learning in deep neural networks","venue":null,"work_id":"32c3e351-2275-472f-add0-3ba99e279125","year":2020},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.081422Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:18a22e16edbc5d89598f3331ed4c5c4b8ca70e2b322587d964a685986afb4876","observation_id":"504dc068-fbb3-4026-b1d4-87f3c4e9d131","resolution":{"observed_at":"2026-08-10T20:08:47.355708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01832","last_updated":"2024-07-18T10:21:29Z","snapshot_observed_at":"2026-08-04T15:45:57.991866Z","submitted_at":"2024-02-02T18:59:58Z","title":"SynthCLIP: Are We Ready for a Fully Synthetic CLIP Training?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01832","snapshot_observed_at":"2026-08-10T20:08:46.086165Z","title":"Synthclip: Are we ready for a fully synthetic clip training? arXiv preprint arXiv:2402.01832, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.086165Z"},"links":{"cited_paper":"/paper/2402.01832","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f2fb4155f2e23131e2dfd83e8dff1206147c3415398924a869157fd21ffaa22b","observation_id":"6ad3e1d3-2a2b-4509-be2f-5b1e12a3db5e","resolution":{"observed_at":"2026-08-10T20:08:46.086165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.329550Z","title":null,"venue":null,"work_id":"fd704792-6614-4823-adca-ca5400697b7f","year":1989},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.091030Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:0f09c8ee6ef82a2c0e3187792079f2059b7909d4e1110506fad4acb5e18e2e7c","observation_id":"863bde2d-1a1e-433d-bfcd-09a3e99ecd6e","resolution":{"observed_at":"2026-08-10T20:08:47.335828Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.308277Z","title":"Quilt-1m: One million image-text pairs for histopathology","venue":null,"work_id":"784cb7cf-c802-41e9-b550-7ac75826b6c9","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.096959Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:0824b60569d92285995dee52b6db84b35ec3a68a4d8a443fb91c79ad04561f2b","observation_id":"fd484441-67ad-4796-b2bf-0cc61cb951c7","resolution":{"observed_at":"2026-08-10T20:08:47.314683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.289449Z","title":"Chexpert: A large chest radiograph dataset with uncertainty labels and expert comparison","venue":null,"work_id":"4f59563c-f337-4a6d-b0b7-89b8b8a5172a","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.103106Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:47f69d4344b62b4715e6e9e8e960327b06b42c8e0db3ff5a504a882a076273e3","observation_id":"73531a14-883c-4bee-842c-7c0fc3a33525","resolution":{"observed_at":"2026-08-10T20:08:47.296473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.270647Z","title":"Generative models as a data source for multiview representa- tion learning","venue":null,"work_id":"579643ec-1ba6-42c6-bcd6-3acc9573ee71","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.108782Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f8fcfa3eab9af04342ea70d5d0c8c507864f7d4c0721fa108f96dd92b46132f2","observation_id":"a0594ace-e525-4638-b328-70feab20beaa","resolution":{"observed_at":"2026-08-10T20:08:47.277193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.251670Z","title":"The power of negation in english: Text, context and relevance","venue":null,"work_id":"b6f4a73b-1178-41cd-8edc-41c20dad13cf","year":1998},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.113421Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:ef1c6d7c2c61bde12118b100291ede4efa5aacb602c667c7585d0834e067051b","observation_id":"6825bbb5-161f-4b27-8c15-85c4acb02932","resolution":{"observed_at":"2026-08-10T20:08:47.257383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.228306Z","title":"Negation in syntax–on the na- ture of functional categories and projections","venue":null,"work_id":"7f096a3a-092a-47d4-80a3-9c93a340da10","year":1990},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.118134Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:58def4e1e794dd54b58565d0295d861e7913083cd2bc42729549f0330cf5b0ce","observation_id":"b40567b7-2ebc-44d1-a989-c1c016b6515d","resolution":{"observed_at":"2026-08-10T20:08:47.234939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.208449Z","title":"Naturalbench: Evalu- ating vision-language models on natural adversarial samples","venue":null,"work_id":"f659f5a2-691c-4571-a68b-b61bf4689a5d","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.122815Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:ad1179d8d099c05da3789f4dfae5da60c1a9c0588068acc90f776beab649bd77","observation_id":"e4b8857b-2a24-492a-8b01-221b1f34db22","resolution":{"observed_at":"2026-08-10T20:08:47.214576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.187847Z","title":"Compre- hending and ordering semantics for image captioning","venue":null,"work_id":"53389f50-f88b-4364-8ebc-9daf9aca9c42","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.128486Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9112d320df4554c05e7f0e27a987a82f1b43cfdd2a344ccf8c9e5f7c8f5fb2ae","observation_id":"51eb013c-e984-4379-8ec6-8ce2bcc29324","resolution":{"observed_at":"2026-08-10T20:08:47.194283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.168069Z","title":"Cross-modal retrieval and semantic re- finement for remote sensing image captioning","venue":null,"work_id":"64ef388b-d927-49f7-8c54-c25ba2e81f5c","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.133410Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:fe3ba2d63363a2a91de39b6291d923e93aef98e022d62b84870e2463088cb4ca","observation_id":"e2d73bba-3d18-4099-bf07-0007c3bfce36","resolution":{"observed_at":"2026-08-10T20:08:47.174919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.150497Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"7c184874-edbf-42e1-9be5-a02c15bab4ea","year":2014},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.138330Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:23f67b72af3960376bcb0678235fd0a0f8a1fcbddb00c9a56c69753dff09b3f9","observation_id":"9c371f82-ffe9-4c86-b92b-0595593f0e24","resolution":{"observed_at":"2026-08-10T20:08:47.156422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.127762Z","title":"A visual- language foundation model for computational pathology","venue":null,"work_id":"067b4534-14d3-48bc-be4b-fd5e9e953677","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.143444Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:88d0b678cc003d0159a54a4951194fa10cadc374a0a6c667fcad861d31fe2d8c","observation_id":"c376ff10-6159-490e-b8f8-24a60b6a017e","resolution":{"observed_at":"2026-08-10T20:08:47.134814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08860","last_updated":"2021-05-08T08:25:57Z","snapshot_observed_at":"2026-08-10T14:43:16.360554Z","submitted_at":"2021-04-18T13:59:50Z","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08860","snapshot_observed_at":"2026-08-10T20:08:46.149164Z","title":"CLIP4Clip: An empirical study of clip for end to end video clip retrieval","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.149164Z"},"links":{"cited_paper":"/paper/2104.08860","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:a829d67048e5533437c179be1e1239f0bb59db1f4fe4a67f6cc14ce35a0c0b56","observation_id":"92c26dc6-4b83-4553-a6ca-f3662ab335d6","resolution":{"observed_at":"2026-08-10T20:08:46.149164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.099166Z","title":"Fine-tuning llama for multi-stage text retrieval","venue":null,"work_id":"4b14f085-b502-44b0-9634-12a17b42a58c","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.154599Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:a263196ecba75fc30600c7efa574f1efed8ab1faf0875d47adfd372374597dbc","observation_id":"46d0c712-e0cf-4ceb-9a8a-be0410d8928e","resolution":{"observed_at":"2026-08-10T20:08:47.105551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.061659Z","title":"Crepe: Can vision-language foundation models reason compositionally? In CVPR, 2023","venue":null,"work_id":"2a445046-a557-42a3-b494-8d3e525d3362","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.159371Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:3100fd884d5c5d58a2c227766662808dca36860fba17805acc0bf6df4e81e431","observation_id":"e298a188-b5e0-4d97-b981-26f8a39358a5","resolution":{"observed_at":"2026-08-10T20:08:47.077283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.040106Z","title":"Simple open-vocabulary object detection with vi- sion transformers","venue":null,"work_id":"26bf24de-69f0-4c34-ab97-179449e9decb","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.164498Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:92fb403c374788e9fda5ff8a35007d526d0aeec555a9f80f123f194b254a2c8b","observation_id":"367d25f1-9eed-41c0-bc92-6b3d86ddcd67","resolution":{"observed_at":"2026-08-10T20:08:47.046531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.015446Z","title":"Recent advances in processing negation","venue":null,"work_id":"09847457-4797-460c-9b1c-03f193eaf114","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.169611Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:40b118587caffb1d82d63a8a3525e21171358144a8e6daf8b50f5e3a4567178f","observation_id":"7bb530a6-5b4e-447e-983a-dc3ff370dd56","resolution":{"observed_at":"2026-08-10T20:08:47.021702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.982419Z","title":"Effect of negation in sentences on sentiment analy- sis and polarity detection","venue":null,"work_id":"3666c431-3ff9-438b-a613-a51c8a2cef3e","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.174089Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:341a9302f0698225e858b0afb093bc6f16b46f205aa6c89ce94d612533d51f7e","observation_id":"98baac9d-107f-4d94-9a8e-fc3102221a5b","resolution":{"observed_at":"2026-08-10T20:08:46.990672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.961213Z","title":"Clip-it! language-guided video summarization","venue":null,"work_id":"235c0f93-d366-45c8-aee9-59de81656a6b","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.178920Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:34b19b1eb26c9f5390e99a3eaa1ed300e21b8519c6aac8365e134e6369e56335","observation_id":"1e9c7f92-9608-4b9a-bbb0-8f78ca2618b7","resolution":{"observed_at":"2026-08-10T20:08:46.967778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.14424","last_updated":"2019-10-31T12:45:40Z","snapshot_observed_at":"2026-08-09T22:20:54.188721Z","submitted_at":"2019-10-31T12:45:40Z","title":"Multi-Stage Document Ranking with BERT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.14424","snapshot_observed_at":"2026-08-10T20:08:46.185183Z","title":"Multi-stage document ranking with bert","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.185183Z"},"links":{"cited_paper":"/paper/1910.14424","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:62c736ec9c1fe83ad63ceb9d1b1fafa5a080fd6337cadabb78dd33fc93c2a4ce","observation_id":"8835517b-641d-4731-9c1e-a04012018817","resolution":{"observed_at":"2026-08-10T20:08:46.185183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.940262Z","title":"Synthesize diagnose and optimize: Towards fine- grained vision-language understanding","venue":null,"work_id":"a86dc088-a867-40df-8d62-2901acf97bff","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.190882Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:220527bff5be9e86dcadf52f32a13ca7c14e736853788b475fcb1d5bd8e49ac4","observation_id":"5869f8b3-04a3-4e44-a339-5be9332767af","resolution":{"observed_at":"2026-08-10T20:08:46.947397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.920315Z","title":"On guiding vi- sual attention with language specification","venue":null,"work_id":"06b6325d-65aa-41f5-95be-6a950447c8fe","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.198307Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:c73b2f6215ba02a6167bc8e0d9b7b9210d739b5ea3bb2e8bafbb5a599f261bfb","observation_id":"3e17222d-e85e-4157-8b69-7f9b531211b2","resolution":{"observed_at":"2026-08-10T20:08:46.926944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.903520Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"142d36a8-8493-46bc-9bd9-319fbc1e426d","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.203619Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:671f75db9e5ab0f21ef9a2aa2f0e19369354c177a1d05e8764f97c165e9f21b1","observation_id":"877627b6-c6ca-4b13-89d2-df8a6aa91d57","resolution":{"observed_at":"2026-08-10T20:08:46.909588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.886237Z","title":"Denseclip: Language-guided dense prediction with context- aware prompting","venue":null,"work_id":"a51d9a37-74ed-4e30-aa28-874cb69c5419","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.208659Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:11f2960728ed823560efb1019ae0cb504d35696d1d96559ef24b7d2c5d3e43e0","observation_id":"1a058905-4956-4332-b0bb-e67c043cadab","resolution":{"observed_at":"2026-08-10T20:08:46.891949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.868007Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":"19d659e0-95cb-4396-912e-f03f2db66a08","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.214930Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9d71b2847df14c98434f1305f1c6c13f1e836cd1f5b87b0ed156f454437dd32b","observation_id":"b2cf1896-10f9-4896-a3a8-551b340ce5b3","resolution":{"observed_at":"2026-08-10T20:08:46.875485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.850248Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"2eab8cd7-60be-4627-baf2-7db994f1f4c2","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.221528Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:cd496f83837ded948242fa70f46bc4c6f6aca3b053e785554c91fa70105c04d0","observation_id":"986b4a43-0592-49e9-b54d-5fdb299a5c7d","resolution":{"observed_at":"2026-08-10T20:08:46.855890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.831533Z","title":"Clip for all things zero-shot sketch-based image retrieval, fine- grained or not","venue":null,"work_id":"17b65315-7408-4575-a0d1-a868bd98e432","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.226461Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:359e08601e2320388fbd43bab8549338b49da8ee6edddc082f36298c78cad02e","observation_id":"850b4137-6f56-4bb0-8303-dfb1b0199aa7","resolution":{"observed_at":"2026-08-10T20:08:46.837669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.792596Z","title":"LAION-5b: An open large-scale dataset for train- ing next generation image-text models","venue":null,"work_id":"a84f676d-7da3-4d19-a565-3c7d122f28b5","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.231797Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f8f36b7f073e6bcaf757c7b0bedd0696708667d863e73e7e0bcfa427c1fdf1bb","observation_id":"1a841611-2507-45f8-b381-b5259404120b","resolution":{"observed_at":"2026-08-10T20:08:46.819673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.773321Z","title":"How much can clip benefit vision-and-language tasks? In International Conference on Learning Representa- tions","venue":null,"work_id":"9de1d9fc-9a25-4513-aba7-c758a00f5235","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.238053Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:e5a2d0f80f433ce00517443ed452ed147db0541fa26e143027cc88e994bab772","observation_id":"b1117bc4-936e-443f-b7f3-fcd2367020e7","resolution":{"observed_at":"2026-08-10T20:08:46.779307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.751904Z","title":"Proposalclip: Unsupervised open-category object pro- posal generation via exploiting clip cues","venue":null,"work_id":"23d167a7-8ae0-47d6-825e-e43708af1178","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.244630Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:4717c9f7f1d07016cfe30af82b00bd61b46d80d4e0f255f1a57215b2fbdb40d8","observation_id":"89029596-1604-4d2e-8baa-02b6076886a1","resolution":{"observed_at":"2026-08-10T20:08:46.759276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.732951Z","title":"Cliport: What and where pathways for robotic manipulation","venue":null,"work_id":"caaf6ed3-5c36-4b2c-9a0a-41bad9752c61","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.250104Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:4e2673587baad7576bf4e41f2783f0b76163d988293c7c8fac218cfb42d317e1","observation_id":"15f7bafa-b741-49ac-964c-95d392e8eba8","resolution":{"observed_at":"2026-08-10T20:08:46.739522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20312","last_updated":"2024-03-29T17:33:42Z","snapshot_observed_at":"2026-08-10T14:28:10.363083Z","submitted_at":"2024-03-29T17:33:42Z","title":"Learn \"No\" to Say \"Yes\" Better: Improving Vision-Language Models via Negations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20312","snapshot_observed_at":"2026-08-10T20:08:46.254987Z","title":"Learn” no” to say” yes” bet- ter: Improving vision-language models via negations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.254987Z"},"links":{"cited_paper":"/paper/2403.20312","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:e83aeebfffd57ec2e9fe6e57e6032d8f155d3243ca97df05d52b664ca60d558a","observation_id":"6b9ddda6-4d36-47c2-8059-4939e8fb9485","resolution":{"observed_at":"2026-08-10T20:08:46.254987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.689611Z","title":"Stablerep: Synthetic images from text-to- image models make strong visual representation learners","venue":null,"work_id":"29ac4242-e750-4e52-968b-2b36fb0b0355","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.260007Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:595cacdbafcec29f72c9ef05763061fa49b6812dffe4d63b374a7b9497f759c9","observation_id":"e6d9dcd2-4121-4365-b480-4979e54b9d75","resolution":{"observed_at":"2026-08-10T20:08:46.707901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.664764Z","title":"Learning vision from mod- els rivals learning vision from data","venue":null,"work_id":"1a0ed6c9-8943-4efc-a520-fc6830364198","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.265791Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9627d8615c64a6880c276f2262be1979546ede17e608233761a4598d145a552d","observation_id":"3cd798c9-eaeb-4969-820e-39ad73b97f21","resolution":{"observed_at":"2026-08-10T20:08:46.670835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.646014Z","title":"Expert-level detection of pathologies from unannotated chest x-ray images via self- supervised learning","venue":null,"work_id":"a5b710a5-b9b0-430d-9797-bb0f73d238db","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.271490Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:95f088a51788c4cc5a488040b908ba2dc4ab5803dee6fb80f6f7d406510cbf41","observation_id":"bacd2688-c153-4551-9cd1-9d3ea0705ac0","resolution":{"observed_at":"2026-08-10T20:08:46.653013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.630506Z","title":"Language models are not naysayers: an anal- ysis of language models on negation benchmarks","venue":null,"work_id":"dbad2b9b-c631-4a2e-b45b-c1cfb3f88f20","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.277242Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:b69200c0519c77b151a2a215151fea56fda60383451e720fd2a558b39bfaf557","observation_id":"505de08f-ebd5-4b72-bb76-e06e2100f58f","resolution":{"observed_at":"2026-08-10T20:08:46.635529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.283190Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.283190Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:72db96c41222efa43278d7a903fcf268e0b483565b64ad018a8e5f6c450bf29b","observation_id":"57d0e0a7-a39f-419a-8995-ab824dc0e9ca","resolution":{"observed_at":"2026-08-10T20:08:46.283190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.601358Z","title":"Real-fake: Effective training data synthesis through distribution matching","venue":null,"work_id":"40b9cc11-3aec-4ae1-b01e-9cec90a1a20a","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.289143Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:82e9fdb5bc692f28fd946511a05c92b8f09e1c64d28a0fffeaa815bb224d0a79","observation_id":"134bce19-b0fc-4a7e-8e3a-d53033da6fed","resolution":{"observed_at":"2026-08-10T20:08:46.607105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.582301Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? In ICLR, 2023","venue":null,"work_id":"db3346b7-98cf-4d93-8160-ba41baf00dd8","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.294239Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:5bbb9ec76d0f5d75801ab69142cca5be1a313144152d67e0b4f6df94c0ca7580","observation_id":"28592488-7af3-4134-a2ea-ce92e7c9575c","resolution":{"observed_at":"2026-08-10T20:08:46.588431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.564560Z","title":"Lit: Zero-shot transfer with locked-image text tuning","venue":null,"work_id":"2919fd67-20c7-4656-aa37-29e7d393ccfc","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.298798Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f6f45808e829ff326507828d914a594e54148c78a2b11e9cab4df94546eef1b3","observation_id":"56c7ca48-e364-4fab-afa8-8f56e3500ae9","resolution":{"observed_at":"2026-08-10T20:08:46.569731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.546392Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"1256601a-06c7-45ff-8811-065ed28acd2a","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.304481Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:eddeec9f9c2d0260277fad1a399ee1585a3b5e7496e36a76b1ce8961308ca7eb","observation_id":"3b85b920-c49b-45cb-b1c9-3d6d2ee36ec1","resolution":{"observed_at":"2026-08-10T20:08:46.552358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.00915","last_updated":"2025-01-08T22:58:51Z","snapshot_observed_at":"2026-07-06T14:57:39.647497Z","submitted_at":"2023-03-02T02:20:04Z","title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.00915","snapshot_observed_at":"2026-08-10T20:08:46.310828Z","title":"Biomedclip: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.310828Z"},"links":{"cited_paper":"/paper/2303.00915","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f37631ca1a198e0bf097e50f375d6bc0d76fb759f16fb5511176a60cd2e58444","observation_id":"5bdf1839-48ec-41c2-8e4d-f6c92c722db5","resolution":{"observed_at":"2026-08-10T20:08:46.310828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.511879Z","title":null,"venue":null,"work_id":"fe47149e-856c-4123-9e9b-0c3cdf0ee6da","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.321808Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:8fc04a329b4ee18e15441655c727cb2c00047e49c8e10611ee9ffbec92de60d9","observation_id":"31792e35-f68e-4868-919a-52efb1ade9b3","resolution":{"observed_at":"2026-08-10T20:08:46.517398Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.493976Z","title":"Yes.” over “No","venue":null,"work_id":"852152c7-cfad-4c25-8def-d683040bdf05","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.326666Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9328caef2263cc593ea6c290bb3df9fd563a2c1c67556f45d525fe15f2f14308","observation_id":"eebb94a9-a692-433c-88eb-a37f3f06fbda","resolution":{"observed_at":"2026-08-10T20:08:46.500166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.491708Z","title":null,"venue":null,"work_id":"5999ddac-27e7-44df-9293-3ff311f5ef6c","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.025541Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:b4545b17aca34ae596377818c59b30a1069f92b54c4a5b672042a382d9c40d3f","observation_id":"6ff600fa-2015-4ed7-996a-b3c7c3a277ae","resolution":{"observed_at":"2026-08-10T20:08:47.498368Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.529211Z","title":null,"venue":null,"work_id":"6972a785-f33f-4262-88c2-3dc62034e3e1","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.316277Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:509c52452030eac67dce5d2e932946b0976915c59d424750e5538f04d51cb5b5","observation_id":"1367e721-a6e2-480a-80d9-672a22dd9629","resolution":{"observed_at":"2026-08-10T20:08:46.534065Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":11,"verified_exact":0,"verified_fuzzy":44},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 8 inbound Pith citation observations for arXiv:2501.09425."}