{"as_of":"2026-08-10T12:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:811bed9b1952618acb288cf39bb8dd2c5a808fbb55fc608280f958cefff50eba","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:21:35.161833Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T03:51:55.155521Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:40:07.157025Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"cited_work":{"arxiv_id":"2506.08227","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.08227","snapshot_observed_at":"2026-07-04T19:40:07.157025Z","title":"Zhiqiu Lin, Xinyue Chen, Deepak Pathak, Pengchuan Zhang, and Deva Ramanan","venue":null,"work_id":"69d2fdb8-c85e-46bc-872f-1c799b02503f","year":2025},"citing_paper":{"arxiv_id":"2604.11496","last_updated":"2026-04-16T10:51:43Z","snapshot_observed_at":"2026-08-03T05:45:05.795046Z","submitted_at":"2026-04-13T14:03:18Z","title":"Revisiting Compositionality in Dual-Encoder Vision-Language Models: The Role of Inference","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:38:00.522094Z"},"links":{"cited_paper":"/paper/2506.08227","citing_paper":"/paper/2604.11496"},"observation_digest":"sha256:a4c660997daa5078dcb11af78114456d8d20b5ba12ad4c5c4287f6f4ca272304","observation_id":"773f9160-465e-4d0e-bd18-743ac80e09ce","resolution":{"observed_at":"2026-05-11T08:26:01.787520Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"cited_work":{"arxiv_id":"2506.08227","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.08227","snapshot_observed_at":"2026-07-04T19:40:07.157025Z","title":"Zhiqiu Lin, Xinyue Chen, Deepak Pathak, Pengchuan Zhang, and Deva Ramanan","venue":null,"work_id":"69d2fdb8-c85e-46bc-872f-1c799b02503f","year":2025},"citing_paper":{"arxiv_id":"2606.25432","last_updated":"2026-06-29T19:08:38Z","snapshot_observed_at":"2026-08-06T21:03:12.899914Z","submitted_at":"2026-06-24T05:50:28Z","title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-06-25T21:05:36.836361Z"},"links":{"cited_paper":"/paper/2506.08227","citing_paper":"/paper/2606.25432"},"observation_digest":"sha256:a2eb0f19fedc0a05d558883121792391538b46bee74608b007b8c81393241266","observation_id":"6547264c-3fe5-413d-aa13-6fc142f74696","resolution":{"observed_at":"2026-07-04T19:40:07.158536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.08227","snapshot_observed_at":"2026-08-01T03:51:55.155521Z","title":"A good CREPE needs more than just Sugar : Investigating biases in compositional vision-language benchmarks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23052","last_updated":"2026-07-25T05:39:51Z","snapshot_observed_at":"2026-08-10T11:17:07.959490Z","submitted_at":"2026-07-25T05:39:51Z","title":"Similarity Is Not Logic: Factored Inference for Dual-Encoder Vision-Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-01T03:51:55.155521Z"},"links":{"cited_paper":"/paper/2506.08227","citing_paper":"/paper/2607.23052"},"observation_digest":"sha256:2ff8cf819d7b8245b02b21f7c51a07b115f59586c86176346955c75c8f994102","observation_id":"a79a2f97-e8b3-4ba6-8924-74995bc2423d","resolution":{"observed_at":"2026-08-01T03:51:55.155521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.08227/citation-record","integrity":"/paper/2506.08227/integrity","json":"/paper/2506.08227/citation-record.json","paper":"/paper/2506.08227"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1811.05013","last_updated":"2018-11-12T21:45:41Z","snapshot_observed_at":"2026-08-10T11:16:44.359362Z","submitted_at":"2018-11-12T21:45:41Z","title":"Blindfold Baselines for Embodied QA","version":1},"cited_work":{"arxiv_id":"1811.05013","doi":null,"metadata_source":"pith","pith_arxiv_id":"1811.05013","snapshot_observed_at":"2026-08-07T05:21:35.577953Z","title":"Blindfold Baselines for Embodied QA","venue":"cs.CV","work_id":"4acbd1d3-c2df-41d4-98ea-945db5672174","year":2018},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.631870Z"},"links":{"cited_paper":"/paper/1811.05013","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:fc8954f1cae1376c9ecac1602edcdd9d5d54b9b3e3f96af1a1653573db051a7e","observation_id":"1315127a-eb00-4076-a473-6c4635f35f22","resolution":{"observed_at":"2026-08-07T05:21:35.582772Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16772","last_updated":"2025-01-22T17:42:29Z","snapshot_observed_at":"2026-08-07T10:39:37.836057Z","submitted_at":"2024-07-23T18:10:43Z","title":"VisMin: Visual Minimal-Change Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16772","snapshot_observed_at":"2026-08-07T05:21:34.697115Z","title":"Vismin: Visual minimal-change understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.697115Z"},"links":{"cited_paper":"/paper/2407.16772","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:605bcee69c7bd3b74a959d35e27ce57d30831e8a973803ed4c7d7f403a5e9756","observation_id":"514ba2d7-166e-42c1-aa13-35a90f453105","resolution":{"observed_at":"2026-08-07T05:21:34.697115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01389","last_updated":"2025-07-14T02:48:47Z","snapshot_observed_at":"2026-08-10T11:16:38.948313Z","submitted_at":"2024-09-02T17:39:26Z","title":"CV-Probes: Studying the interplay of lexical and world knowledge in visually grounded verb understanding","version":2},"cited_work":{"arxiv_id":"2409.01389","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01389","snapshot_observed_at":"2026-08-07T05:21:35.539827Z","title":"CV-Probes: Studying the interplay of lexical and world knowledge in visually grounded verb understanding","venue":"cs.CL","work_id":"eae46629-4c7b-459b-aa33-47034c0a1130","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.839756Z"},"links":{"cited_paper":"/paper/2409.01389","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:40e604eacb75934885378cc7cbcd7042f6b106c34950f55922b0c1e4de36aa94","observation_id":"ff43a8e0-0a48-4bd9-9b0b-7175df3f170c","resolution":{"observed_at":"2026-08-07T05:21:35.545077Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.941439Z","title":"Evil- probe-a composite benchmark for extensive visio-linguistic probing","venue":null,"work_id":"39d1e18a-5453-432e-b9d0-dcd6acb60150","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.925092Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:fe7c5422eb6020dd343160a9fbefc2849bfa62f4a5445a3c3e9953b6110402c7","observation_id":"e456ef2e-6b3c-4028-9530-ef8c92b00ed7","resolution":{"observed_at":"2026-08-07T05:21:35.945624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04492","last_updated":"2024-08-06T17:31:33Z","snapshot_observed_at":"2026-08-08T13:12:38.339091Z","submitted_at":"2024-02-07T00:31:49Z","title":"ColorSwap: A Color and Word Order Dataset for Multimodal Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04492","snapshot_observed_at":"2026-08-07T05:21:34.962463Z","title":"Colorswap: A color and word order dataset for mul- timodal evaluation.arXiv preprint arXiv:2402.04492, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.962463Z"},"links":{"cited_paper":"/paper/2402.04492","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:d6a6b401008562f5fd7237cd1306bfe97427ff028410db34f30adcc444f79c9b","observation_id":"6f3b9892-4235-476c-92a3-98bac5295db2","resolution":{"observed_at":"2026-08-07T05:21:34.962463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15021","last_updated":"2024-03-01T01:52:58Z","snapshot_observed_at":"2026-08-10T11:18:08.804816Z","submitted_at":"2024-02-22T23:42:25Z","title":"CLoVe: Encoding Compositional Language in Contrastive Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15021","snapshot_observed_at":"2026-08-07T05:21:34.967677Z","title":"Clove: Encoding compositional lan- guage in contrastive vision-language models.arXiv preprint arXiv:2402.15021, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.967677Z"},"links":{"cited_paper":"/paper/2402.15021","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:ab8ce431f89caa29f492cea73fad84dbca5d085afec9b288a9ea70bf6f58183f","observation_id":"8c9aacce-3096-449c-9206-cc24b38c8c1f","resolution":{"observed_at":"2026-08-07T05:21:34.967677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15061","last_updated":"2023-10-23T16:05:13Z","snapshot_observed_at":"2026-08-08T13:46:11.586887Z","submitted_at":"2023-10-23T16:05:13Z","title":"The BLA Benchmark: Investigating Basic Language Abilities of Pre-Trained Multimodal Models","version":1},"cited_work":{"arxiv_id":"2310.15061","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.15061","snapshot_observed_at":"2026-08-07T05:21:35.491611Z","title":"The BLA Benchmark: Investigating Basic Language Abilities of Pre-Trained Multimodal Models","venue":"cs.CL","work_id":"e725221d-ef23-4aca-85b3-e9749ce49a8a","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.972843Z"},"links":{"cited_paper":"/paper/2310.15061","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:ba07049ec5fa0542d3a5b79013a468d4f512d85a29de3819c9f0b2884bbdb61b","observation_id":"4a5409ad-f390-42cf-80ed-30fddc6f4908","resolution":{"observed_at":"2026-08-07T05:21:35.495986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.927887Z","title":"Routledge, 2016","venue":null,"work_id":"1abf3254-fa66-4058-b9b7-a8054d9f5b8b","year":2016},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.977964Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:9abcd807dddcd42fe0bbc3fc6e6cba9e4943484d8ddea02b34120218a6489d96","observation_id":"e06e66ca-e87f-4833-a575-c135a8440cab","resolution":{"observed_at":"2026-08-07T05:21:35.932358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.914031Z","title":"Sugarcrepe++ dataset: Vision-language model sensitivity to semantic and lexical alterations.Advances in Neural Information Processing Systems, 2024","venue":null,"work_id":"32d7bb16-dfa0-4f0b-9c95-038de8f09233","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.982501Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:384352be4673476046982df44801a763b9c1ba52e46963efda4ef5e6f8cf5c9f","observation_id":"38918694-0f76-466d-950b-0770cb460411","resolution":{"observed_at":"2026-08-07T05:21:35.918635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.900692Z","title":"Dat- acomp: In search of the next generation of multimodal datasets.Advances in Neural Information Processing Sys- tems, 36:27092–27112, 2023","venue":null,"work_id":"37a215ca-9e76-4fb7-baf6-27cce6f2a39c","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.986765Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:6f926ec4f5804c27767d2c13230dfc695c30ee17211f97a17e66ceab6f65a1a5","observation_id":"99754130-9e4a-4f79-ab05-62c52a5a0c9e","resolution":{"observed_at":"2026-08-07T05:21:35.904909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.886983Z","title":"Shortcut learning in deep neural networks","venue":null,"work_id":"15d0c587-8ef9-48a1-82a1-92deba74466a","year":2020},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.990808Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:2715a377ab73a6b48f2d6aca714aeb52620cddb84bccebb5f01ad12b7d2ae16c","observation_id":"7a66031c-fb9e-4c71-a5f1-ff2f5e9d7c3c","resolution":{"observed_at":"2026-08-07T05:21:35.891192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:34.995102Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answer- ing","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.995102Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:a25d50f105841507e2f92ec5d82b0001500dc6c15d7a564dec18b98bbe6304c5","observation_id":"d2731f8b-0024-4f47-890d-68ad2b604ddc","resolution":{"observed_at":"2026-08-07T05:21:34.995102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.865266Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning","venue":null,"work_id":"e98b2441-6a49-4304-9a9e-e1e4dc880d29","year":2021},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:34.999500Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:410ae67881d7831f6a7f5986576b2c1f26bd48b98097d2efd1bb822544fe6cbb","observation_id":"1136e946-f1a2-43d0-af6f-40eaf60bf8c0","resolution":{"observed_at":"2026-08-07T05:21:35.869530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09141","last_updated":"2021-06-16T21:36:36Z","snapshot_observed_at":"2026-08-10T03:10:05.954912Z","submitted_at":"2021-06-16T21:36:36Z","title":"Probing Image-Language Transformers for Verb Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09141","snapshot_observed_at":"2026-08-07T05:21:35.003872Z","title":"Probing image-language transformers for verb understanding.arXiv preprint arXiv:2106.09141, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.003872Z"},"links":{"cited_paper":"/paper/2106.09141","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:b4bb6cd39b3a428434b064c64e8bb80f39e551ab331888d12836d052559151bc","observation_id":"177fad96-974e-474e-9b39-9f75b9d9a461","resolution":{"observed_at":"2026-08-07T05:21:35.003872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.851842Z","title":"Sugarcrepe: Fixing hackable benchmarks for vision-language compositionality.NeurIPS,","venue":null,"work_id":"51c7b8df-667e-432b-b2eb-bbed9a353d3e","year":null},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.008007Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:f1cefd73726bf112738c33984aa5cb3bd7aed9c1e24612a013667b4e0ca5c8c1","observation_id":"4db1d596-d0b0-4a0f-b451-e56cfff7a7c1","resolution":{"observed_at":"2026-08-07T05:21:35.855961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.03067","last_updated":"2018-04-24T10:25:07Z","snapshot_observed_at":"2026-07-06T06:27:14.996782Z","submitted_at":"2018-03-08T12:37:14Z","title":"Compositional Attention Networks for Machine Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.03067","snapshot_observed_at":"2026-08-07T05:21:35.012585Z","title":"Compositional attention networks for machine reasoning.arXiv preprint arXiv:1803.03067, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.012585Z"},"links":{"cited_paper":"/paper/1803.03067","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:4b4555973c63716d51abbe16b2dff28c3561fbe39ed9095a8c096c248e892dc7","observation_id":"5eb6ce73-4dc4-43e1-978c-e826b9d3dfdd","resolution":{"observed_at":"2026-08-07T05:21:35.012585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.838997Z","title":"Text encoders bottleneck compositionality in contrastive vision- language models","venue":null,"work_id":"e28d704b-9da9-446a-80c4-cbac1f390f8b","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.016845Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:b2f665b1343b014998fa5c6ac13e403969f9aad0d369f8b6b356840894787ca0","observation_id":"005709c4-5903-410d-8479-eb5772704c0e","resolution":{"observed_at":"2026-08-07T05:21:35.842984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19785","last_updated":"2023-10-30T17:50:15Z","snapshot_observed_at":"2026-08-10T11:15:55.769939Z","submitted_at":"2023-10-30T17:50:15Z","title":"What's \"up\" with vision-language models? Investigating their struggle with spatial reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19785","snapshot_observed_at":"2026-08-07T05:21:35.021243Z","title":"What’s” up” with vision-language models? investigating their strug- gle with spatial reasoning.arXiv preprint arXiv:2310.19785,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.021243Z"},"links":{"cited_paper":"/paper/2310.19785","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:27cc3deed623443717e167d4a9824bb159a4fa4a69541b8a38e3cacd542f2d18","observation_id":"14cc2461-dbfe-4320-a637-c8ec647f4f39","resolution":{"observed_at":"2026-08-07T05:21:35.021243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.826008Z","title":"The hard positive truth about vision-language compositionality","venue":null,"work_id":"006de2a8-ea6b-4d79-b36a-e13415d137e4","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.026678Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:7612de0eb776068e6e6023e44b6ef2e77ec99f00969a37fd88422a86363c5d6a","observation_id":"ebc2b2ef-b1fb-4778-aa38-06d7686a4dce","resolution":{"observed_at":"2026-08-07T05:21:35.830314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.030959Z","title":"Clip behaves like a bag-of-words model cross-modally but not uni-modally.arXiv preprint arXiv:2502.03566, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.030959Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:7324348a4abc0fd22f3ee860974723fe89fe1f988359c500dc3025b55b8d4d87","observation_id":"9c5d8276-f12b-47e7-8b59-3f0301333ac8","resolution":{"observed_at":"2026-08-07T05:21:35.030959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.035604Z","title":"Building machines that learn and think like people.Behavioral and brain sciences, 40:e253,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.035604Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:58eda327e20f8c70f1301786f365b0558ccb068da34ff9af4c54c56e9e4956fb","observation_id":"89abbafb-e363-47cb-9007-e6bbc3e3731a","resolution":{"observed_at":"2026-08-07T05:21:35.035604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.804757Z","title":"Coco- counterfactuals: Automatically constructed counterfactual examples for image-text pairs.Advances in Neural Infor- mation Processing Systems, 2023","venue":null,"work_id":"5175761a-5fe7-48b9-9fca-5fdeadf3925f","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.039797Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:7d21dfa69f53a31f37353540c38205ad383a74c62ee7eda91e112d82d55c9364","observation_id":"458b3f24-8cbf-4a93-8b7d-60019b2fd7a3","resolution":{"observed_at":"2026-08-07T05:21:35.808751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01167","last_updated":"2025-03-29T09:39:11Z","snapshot_observed_at":"2026-08-07T17:34:54.910235Z","submitted_at":"2025-03-03T04:30:39Z","title":"Enhancing Vision-Language Compositional Understanding with Multimodal Synthetic Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01167","snapshot_observed_at":"2026-08-07T05:21:35.043702Z","title":"Enhancing vision-language com- positional understanding with multimodal synthetic data","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.043702Z"},"links":{"cited_paper":"/paper/2503.01167","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:32eadce7ab5e3243cf33e8986aaa3ca6aee83034767aaeaf8aaeb5f5baddeaf9","observation_id":"93a25d74-a6cf-4136-8f00-08f039e07271","resolution":{"observed_at":"2026-08-07T05:21:35.043702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.791380Z","title":"Remov- ing distributional discrepancies in captions improves image- text alignment","venue":null,"work_id":"7bc6ca79-bd63-4524-bf28-b11dc5480f55","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.048382Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:3826e5076e7e4b422e6e02070683df2608eaac74d32064c2fca560af35fc5cc4","observation_id":"72da09fc-fc4c-44b3-87e4-9ccfdc15f9fd","resolution":{"observed_at":"2026-08-07T05:21:35.795629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.777833Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"ffeef6d8-2821-4b82-9058-cdb0013c7c0e","year":2014},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.053390Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:2033be3c3d5bd2276cd4ed1a0672d6233ac0f4d971b3cef754b079c9b015cb80","observation_id":"dee2741e-8289-4c75-98ec-5d5620aa12a9","resolution":{"observed_at":"2026-08-07T05:21:35.781736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.765451Z","title":"Vera: A general- purpose plausibility estimation model for commonsense statements","venue":null,"work_id":"f04cbda3-2236-4b46-bfa1-ac8d485bb671","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.057677Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:6acb42312b08cbcbfd9a535eb9d840204871d148df2eab6324d6fa7873581a1e","observation_id":"ae714ec9-4090-46e2-923a-b3a46c576065","resolution":{"observed_at":"2026-08-07T05:21:35.769409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.751672Z","title":"Crepe: Can vision-language foundation models reason compositionally? InCVPR, 2023","venue":null,"work_id":"f7fa0818-6ef4-463f-96e8-31f6261f8e77","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.061946Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:8f3072d2d3679fd1bf6303c91a1ff47bb63ed080bafb3b50652c3a2d91bbd1ff","observation_id":"aa6a9c88-529b-4294-a2d2-21b848ec309f","resolution":{"observed_at":"2026-08-07T05:21:35.755690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.738891Z","title":"Compositional chain-of-thought prompting for large multimodal models","venue":null,"work_id":"32f8f09c-829d-468c-9a1a-3974448069b0","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.066367Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:7843e4589d98c61fd4414f7443450cfb03b33125b4661290f8736b16640f9d5f","observation_id":"81902300-1f71-49e1-b56c-c86878d31958","resolution":{"observed_at":"2026-08-07T05:21:35.742994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.05909","last_updated":"2020-10-05T00:10:24Z","snapshot_observed_at":"2026-08-05T07:11:54.244493Z","submitted_at":"2020-04-29T21:33:35Z","title":"TextAttack: A Framework for Adversarial Attacks, Data Augmentation, and Adversarial Training in NLP","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.05909","snapshot_observed_at":"2026-08-07T05:21:35.071014Z","title":"Textattack: A framework for adversarial attacks, data augmentation, and adversarial training in nlp","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.071014Z"},"links":{"cited_paper":"/paper/2005.05909","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:451d19284a7f66e82bd606d77f4dcb655f02cdae55d33f19e39308bc4b5b8ecb","observation_id":"6ad48054-2851-4434-8254-b444e1fc6a33","resolution":{"observed_at":"2026-08-07T05:21:35.071014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05210","last_updated":"2024-10-07T17:16:20Z","snapshot_observed_at":"2026-08-10T11:08:22.765241Z","submitted_at":"2024-10-07T17:16:20Z","title":"Preserving Multi-Modal Capabilities of Pre-trained VLMs for Improving Vision-Linguistic Compositionality","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05210","snapshot_observed_at":"2026-08-07T05:21:35.075784Z","title":"Preserving multi-modal capabilities of pre- trained vlms for improving vision-linguistic compositional- ity.arXiv preprint arXiv:2410.05210, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.075784Z"},"links":{"cited_paper":"/paper/2410.05210","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:1a91c227f94c6d9466dd789b3cfb3f2974fc4de2ff459b4d78055063c43c4216","observation_id":"75edd83d-e0b1-49ea-a639-88d5cbf2042a","resolution":{"observed_at":"2026-08-07T05:21:35.075784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.07566","last_updated":"2022-03-14T15:08:08Z","snapshot_observed_at":"2026-08-09T16:57:52.860988Z","submitted_at":"2021-12-14T17:15:04Z","title":"VALSE: A Task-Independent Benchmark for Vision and Language Models Centered on Linguistic Phenomena","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.07566","snapshot_observed_at":"2026-08-07T05:21:35.079871Z","title":"Valse: A task-independent benchmark for vision and language mod- els centered on linguistic phenomena.arXiv preprint arXiv:2112.07566, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.079871Z"},"links":{"cited_paper":"/paper/2112.07566","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:e37e221f4dc21915f2aedc7e52f145d5ed58600a7b9358ae749d359cfce12354","observation_id":"c2d9873b-fb69-4996-a5ea-eeeff6c3f773","resolution":{"observed_at":"2026-08-07T05:21:35.079871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.725559Z","title":"Triplet- clip: Improving compositional reasoning of clip via synthetic vision-language negatives.Advances in Neural Information Processing Systems, 37:32731–32760, 2024","venue":null,"work_id":"37fcc276-0b39-41b4-b2f4-841dd5ecf573","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.084098Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:41d6ce0ed2b59209e8c3b9dc57a0a3d239f2d6a02472a611a8fee646505a9841","observation_id":"f78dec15-7bb9-42b2-a5f5-a716e191633a","resolution":{"observed_at":"2026-08-07T05:21:35.730065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.088354Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.088354Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:dfcc16699f619dea2e24714749398250ce032c33ffb717620306155c13ed44ab","observation_id":"e94f47ec-3ca1-44e7-bfb7-f685dbd980f8","resolution":{"observed_at":"2026-08-07T05:21:35.088354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.703345Z","title":"cola: A bench- mark for compositional text-to-image retrieval.NeurIPS,","venue":null,"work_id":"e4579ed3-e435-40a6-930d-7870c651a0f2","year":null},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.092289Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:23988b1bc793640e1f62b2b6407c840c2a1f534584cc006da9b559392eaa2fc7","observation_id":"50f17bcd-ba9a-48db-97ff-4fa4316e24c9","resolution":{"observed_at":"2026-08-07T05:21:35.707422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11685","last_updated":"2025-01-04T19:33:49Z","snapshot_observed_at":"2026-07-06T18:16:29.163229Z","submitted_at":"2024-05-19T22:04:57Z","title":"ColorFoil: Investigating Color Blindness in Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11685","snapshot_observed_at":"2026-08-07T05:21:35.096863Z","title":"Colorfoil: Investigating color blind- ness in large vision and language models.arXiv preprint arXiv:2405.11685, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.096863Z"},"links":{"cited_paper":"/paper/2405.11685","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:905c44057ca5f02e36b5ba47f4d2cbb6d9c0aea9201bc5acacdaef8a872dfe45","observation_id":"6622fd16-4bca-4cb5-adce-d1f7c9e7cceb","resolution":{"observed_at":"2026-08-07T05:21:35.096863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20312","last_updated":"2024-03-29T17:33:42Z","snapshot_observed_at":"2026-08-07T21:16:08.794114Z","submitted_at":"2024-03-29T17:33:42Z","title":"Learn \"No\" to Say \"Yes\" Better: Improving Vision-Language Models via Negations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20312","snapshot_observed_at":"2026-08-07T05:21:35.100979Z","title":"Learn” no” to say” yes” bet- ter: Improving vision-language models via negations.arXiv preprint arXiv:2403.20312, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.100979Z"},"links":{"cited_paper":"/paper/2403.20312","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:2eeef32bc027fcaa6b9f9a265f79a9cef240d1b32a7ef16a1bad17ad4f88cb17","observation_id":"01b8eed0-e6a3-46b4-9a55-64940aa8889f","resolution":{"observed_at":"2026-08-07T05:21:35.100979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.690126Z","title":"Teaching composition- ality to cnns","venue":null,"work_id":"e65a8cfa-ad32-49c9-8035-e9c1fb3f8ab1","year":null},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.105148Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:6486a4d5bfd5650467425eaa0af496631174e03d1319fc0c41e4e3fb2e40fd19","observation_id":"48eca0ff-40c0-49d5-a724-6e4abdaad982","resolution":{"observed_at":"2026-08-07T05:21:35.694423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.677113Z","title":"Winoground: Probing vision and language models for visio- linguistic compositionality","venue":null,"work_id":"84139a39-eb12-4164-90ff-4a7fa0fe1289","year":2022},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.109185Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:076be74c128c29aa751e6b8f8674c29c136baa6c65389a9c9a37855fe6af41d7","observation_id":"679d7e41-8db7-4673-8be4-4eb0ea8d00b2","resolution":{"observed_at":"2026-08-07T05:21:35.681245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T05:21:35.113327Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms.arXiv preprint arXiv:2401.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.113327Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:ef93c0bd1d447737e761de96e502d626d8be7e5ef01ed441921568a667e05aa1","observation_id":"a3698515-14c1-45aa-a30f-59ac2fb2f97a","resolution":{"observed_at":"2026-08-07T05:21:35.113327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T05:21:35.117743Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.117743Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:d8a467f6fdf6a80e0c77f02b9be0bdfc9f1dec32d1cc92809c62005486fafbfe","observation_id":"b6d1e98b-3f4c-44db-b8a4-a016b56c09a4","resolution":{"observed_at":"2026-08-07T05:21:35.117743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.664325Z","title":"Image captioners are scalable vision learners too.Advances in Neural Infor- mation Processing Systems, 36, 2024","venue":null,"work_id":"ce910c1d-12c3-4b40-b108-7498aa06d68d","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.121739Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:cb5af348e5e48ea5d4e61b232a3d10a1fcaa84fe4dbaf88e31b25ea6800a7bfa","observation_id":"2075b482-a3c3-4074-860f-e7c0d92f2ff9","resolution":{"observed_at":"2026-08-07T05:21:35.668355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.649925Z","title":"Image captioners are scalable vision learners too.Advances in Neural Infor- mation Processing Systems, 36, 2024","venue":null,"work_id":"c620a322-279a-40e7-8e6a-aabe9047c9d9","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.126072Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:de18a32d685b557b0a9b13869e540a896b577117a23bab12bc43f02e9ed4cffb","observation_id":"dbb94afc-f6c8-4b71-bcd1-584bfa3c92e0","resolution":{"observed_at":"2026-08-07T05:21:35.654603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.636509Z","title":"Equivariant similarity for vision-language foundation models","venue":null,"work_id":"3c4c738e-f221-498a-b846-7dbeae447f7e","year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.130067Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:a385114b1acabd231a5bb7f5f3c18545f893c8dae3d8452ad14d1756b65f356f","observation_id":"0ca2d9da-5c56-47dc-badc-5060aaec599a","resolution":{"observed_at":"2026-08-07T05:21:35.640730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10029","last_updated":"2024-12-13T10:39:31Z","snapshot_observed_at":"2026-07-06T20:06:30.732685Z","submitted_at":"2024-12-13T10:39:31Z","title":"Enhancing Fine-Grained Vision-Language Pretraining with Negative Augmented Samples","version":1},"cited_work":{"arxiv_id":"2412.10029","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.10029","snapshot_observed_at":"2026-08-07T05:21:35.243184Z","title":"Enhancing Fine-Grained Vision-Language Pretraining with Negative Augmented Samples","venue":"cs.CV","work_id":"c5fe499a-4d5a-4460-9562-74e64d711361","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.134083Z"},"links":{"cited_paper":"/paper/2412.10029","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:51cf69ed3459cbf1cc77d1d20b41be6fa78bf16dfedd07eccc213f93363fd6b8","observation_id":"4cce305e-31aa-4962-8c5e-eaacd241025a","resolution":{"observed_at":"2026-08-07T05:21:35.249852Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.622956Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? InThe Eleventh International Conference on Learning Representations, 2022","venue":null,"work_id":"a7086f33-8e44-4209-9502-cc8e8b774bd2","year":2022},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.139226Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:a936cf1ffc95571c142401b82a51c1db2088a30a277a81ba678339e08bcf6229","observation_id":"c88e7dc5-7c2d-4a58-8584-5ff64e0d218f","resolution":{"observed_at":"2026-08-07T05:21:35.627745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.608784Z","title":"Investigating compositional chal- lenges in vision-language models for visual grounding","venue":null,"work_id":"335f8434-ee54-4908-99c0-2ef12bb69751","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.143356Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:5f51e5d8125ddcf39a71a961808e01c9f46afe9fe563e0bc8cb7518da4a6053b","observation_id":"bfcb0d5d-3e2b-4b3d-aa1b-a8702e7af27f","resolution":{"observed_at":"2026-08-07T05:21:35.613423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13254","last_updated":"2024-06-12T17:59:55Z","snapshot_observed_at":"2026-07-06T17:33:00.716054Z","submitted_at":"2024-02-20T18:59:55Z","title":"CounterCurate: Enhancing Physical and Semantic Visio-Linguistic Compositional Reasoning via Counterfactual Examples","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13254","snapshot_observed_at":"2026-08-07T05:21:35.148475Z","title":"Countercurate: Enhancing physical and semantic visio- linguistic compositional reasoning via counterfactual exam- ples.arXiv preprint arXiv:2402.13254, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.148475Z"},"links":{"cited_paper":"/paper/2402.13254","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:ebc423d9c02ac4be1bdd5b3c0bcb39f029a8df15294e4105a5569cdedc1424a9","observation_id":"e2f2da20-3329-4311-a1f6-e899b82cfd34","resolution":{"observed_at":"2026-08-07T05:21:35.148475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.08832","last_updated":"2024-04-25T15:24:11Z","snapshot_observed_at":"2026-07-06T15:42:44.954861Z","submitted_at":"2023-06-15T03:26:28Z","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.08832","snapshot_observed_at":"2026-08-07T05:21:35.153214Z","title":"Contrast- ing intra-modal and ranking cross-modal hard negatives to enhance visio-linguistic fine-grained understanding.arXiv preprint arXiv:2306.08832, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.153214Z"},"links":{"cited_paper":"/paper/2306.08832","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:ec99e490e78da8753c6bfc17eb2ddc44acf015c32f9ca80d43e921afeff15dba","observation_id":"8c230ce9-de7e-47c6-9477-98231b1afb40","resolution":{"observed_at":"2026-08-07T05:21:35.153214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.00221","last_updated":"2023-06-22T16:55:44Z","snapshot_observed_at":"2026-08-10T11:59:13.699690Z","submitted_at":"2022-07-01T06:25:53Z","title":"VL-CheckList: Evaluating Pre-trained Vision-Language Models with Objects, Attributes and Relations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.00221","snapshot_observed_at":"2026-08-07T05:21:35.157557Z","title":"Vl- checklist: Evaluating pre-trained vision-language models with objects, attributes and relations.arXiv preprint arXiv:2207.00221, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.157557Z"},"links":{"cited_paper":"/paper/2207.00221","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:2312d19bea07f1d0f5f60f9d5ed45f95dc1bf8c571f18109b695df0d63337641","observation_id":"42821a0d-6d10-4218-a568-1379fe734f45","resolution":{"observed_at":"2026-08-07T05:21:35.157557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:21:35.593776Z","title":"Iterated learning improves composition- ality in large vision-language models","venue":null,"work_id":"62c9f11c-4102-4614-a65a-2cbc9545b26a","year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.161833Z"},"links":{"citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:d0b520221d261774f6bf19b772f5902f85b4488144986976f0e904179dd7ebf7","observation_id":"30cc44a3-8be7-42e3-864c-83382a985605","resolution":{"observed_at":"2026-08-07T05:21:35.598478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T11:16:15.032911Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":4,"verified_fuzzy":25},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 3 inbound Pith citation observations for arXiv:2506.08227."}