{"as_of":"2026-08-10T03:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1d5421ac34ba31b9a5d3d6fbe2313d8a86e619043209cc736616307f8bf158bd","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T00:12:36.209585Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":74,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2307.05663","last_updated":"2023-07-11T17:57:40Z","snapshot_observed_at":"2026-07-06T15:52:52.180267Z","submitted_at":"2023-07-11T17:57:40Z","title":"Objaverse-XL: A Universe of 10M+ 3D Objects","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T13:02:11.512409Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2307.05663"},"observation_digest":"sha256:19bdf6d859d88f4738f825e4c43f2142c30acbfd4f9795307f26b947068cfa98","observation_id":"53e4f70b-9858-4a6d-889c-4aa69b2da6f1","resolution":{"observed_at":"2026-05-17T13:02:11.604349Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:6120063733ddf8cc69e6324d0f2444a1d257b47a923cf539734a618e9e851fa8","observation_id":"5b0763c5-fe42-473d-ae4b-031e297c053a","resolution":{"observed_at":"2026-05-15T06:30:22.712246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T01:52:01.163900Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2308.01390"},"observation_digest":"sha256:15d70b02ff1fd4d1830eb802cb6ded8646dbe13c433bb68ed24abad79e7fbf50","observation_id":"b7853a04-19fe-49b5-bcb6-5fa3b35955f9","resolution":{"observed_at":"2026-05-14T01:52:01.415728Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2311.04257","last_updated":"2023-11-09T01:56:51Z","snapshot_observed_at":"2026-08-06T04:08:15.266897Z","submitted_at":"2023-11-07T14:21:29Z","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-18T03:18:51.582340Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2311.04257"},"observation_digest":"sha256:309ee754fe54424e539d66059df75dcba6d188fffa439b5b223f162a2f96696e","observation_id":"50537838-c88e-44dd-add6-32810a4947d5","resolution":{"observed_at":"2026-05-18T03:18:51.844203Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-04T08:17:54.774738Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T17:08:12.727773Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2311.12793"},"observation_digest":"sha256:1b897f84351ecb9e365d2ed8383d5b81593f2ce4edb81357f6e63b5b714d9e42","observation_id":"dbd967c6-4515-481d-8b9a-4f0905432019","resolution":{"observed_at":"2026-05-13T17:08:12.842273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T22:58:16.523267Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2406.11794"},"observation_digest":"sha256:88ec6650a5af65476eb8a98332dc898d2fcc38d30ac5a84497306b67b3df3056","observation_id":"16a27061-3fdc-4a50-b658-94fc2332560b","resolution":{"observed_at":"2026-05-17T22:58:16.961906Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":211,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ffe36c98d7e9343ad550a7776c86fb34aa02486e506d0b7feb05bbf75104f5f4","observation_id":"26e6ffb9-7845-4953-af59-dcce8109799e","resolution":{"observed_at":"2026-05-20T06:20:36.377313Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-09T00:12:36.209585Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03950","last_updated":"2025-05-18T22:20:47Z","snapshot_observed_at":"2026-08-09T00:05:17.355061Z","submitted_at":"2025-02-06T10:40:42Z","title":"LR0.FM: Low-Res Benchmark and Improving Robustness for Zero-Shot Classification in Foundation Models","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T00:12:36.209585Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2502.03950"},"observation_digest":"sha256:7367825ae9c5c7435e18c1a5b7b24e9c363f652ae9c930e5050824a60e115c85","observation_id":"7dc61a48-3420-4b62-82db-fccc3d0d19ae","resolution":{"observed_at":"2026-08-09T00:12:36.209585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-02T07:57:37.201421Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:179eb0607a4b17d3fd01da8411a49bf549930d4b9882d7e8e1c406cf8f5c2c2d","observation_id":"30ac765f-d39d-43fa-9d45-b57bc71a3f6f","resolution":{"observed_at":"2026-05-22T19:52:01.977529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T15:42:32.345850Z","title":"Y .et al.Datacomp: In search of the next generation of multimodal datasets (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14204","last_updated":"2025-05-20T11:04:14Z","snapshot_observed_at":"2026-08-09T20:36:28.736717Z","submitted_at":"2025-05-20T11:04:14Z","title":"Beginning with You: Perceptual-Initialization Improves Vision-Language Representation and Alignment","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:32.345850Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.14204"},"observation_digest":"sha256:f142c2b5e80c6bc75a4ca41f8a24a7311ca33296c687b366e6df725697ac7dba","observation_id":"aadfe9c0-a28f-4092-b195-fc2eae3e4cd4","resolution":{"observed_at":"2026-08-07T15:42:32.345850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T08:34:36.824053Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:bac254c0f50212fd5c54ffe90ba59f49dd99f58d724709157e0c36bf1ad67e81","observation_id":"e6df2df8-30ed-41f0-b903-f650583f6d3b","resolution":{"observed_at":"2026-05-16T08:34:36.875949Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T00:59:13.826054Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:41d85d2b0ee915d859607460938391d879090db9a2c513822cfad451e3787675","observation_id":"38170498-c1f5-4e46-8004-1602482eff3c","resolution":{"observed_at":"2026-05-22T01:00:51.274211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T11:26:56.445045Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02698","last_updated":"2025-06-06T03:14:26Z","snapshot_observed_at":"2026-08-09T04:25:20.294383Z","submitted_at":"2025-06-03T09:47:22Z","title":"Smoothed Preference Optimization via ReNoise Inversion for Aligning Diffusion Models with Varied Human Preferences","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:56.445045Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.02698"},"observation_digest":"sha256:4237ef5afed16eb688df8fedd156059150871ed7577dffb044cc8efa59ac6aca","observation_id":"fceae8d0-9b28-4509-a9a7-255461e1da22","resolution":{"observed_at":"2026-08-07T11:26:56.445045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T05:27:16.025685Z","title":"Dat- acomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08071","last_updated":"2025-06-09T17:54:41Z","snapshot_observed_at":"2026-08-10T00:01:32.894604Z","submitted_at":"2025-06-09T17:54:41Z","title":"CuRe: Cultural Gaps in the Long Tail of Text-to-Image Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:16.025685Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.08071"},"observation_digest":"sha256:1152e6a07213be97f6898acfd184a931061fc300c4d35f83c7ab5f9b86fd7926","observation_id":"1a413808-640f-485c-a0d7-4112960d33e7","resolution":{"observed_at":"2026-08-07T05:27:16.025685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T04:59:40.265102Z","title":"Gallou´edec, Q., Beeching, E., Romac, C., and Dellandr´ea, E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-09T02:05:58.972748Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.265102Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:bcdddfa945332fd97ba76ebcebce9c3d360fa63717aa42e670960bbfbee5e12d","observation_id":"28da12cc-c86a-4b77-a6b4-f5ab6e92c7cb","resolution":{"observed_at":"2026-08-07T04:59:40.265102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T05:01:13.736890Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10038","last_updated":"2025-06-10T22:37:39Z","snapshot_observed_at":"2026-08-09T04:36:51.228004Z","submitted_at":"2025-06-10T22:37:39Z","title":"Ambient Diffusion Omni: Training Good Models with Bad Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:01:13.736890Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.10038"},"observation_digest":"sha256:fe06ebbac191d870d5b60e6526a3566b0832528bb5713d038b989f447f439986","observation_id":"12352945-800f-405c-90d6-e72bb1f3bac9","resolution":{"observed_at":"2026-08-07T05:01:13.736890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-06T22:03:46.175499Z","title":"Datacomp: In search of the next generation of multimodal datasets.arXiv preprint arXiv:2304.14108, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22881","last_updated":"2026-05-31T13:32:45Z","snapshot_observed_at":"2026-08-06T21:53:10.592056Z","submitted_at":"2025-06-28T13:36:44Z","title":"CLIP-like Model as a Foundational Density Ratio Estimator","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:03:46.175499Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.22881"},"observation_digest":"sha256:8db9d7318abdd3bb688516c1b60d80342c1b23e92e7ff1bf158c919fcb7e0464","observation_id":"8ca22854-6d9e-43e6-a931-eeb949699ba6","resolution":{"observed_at":"2026-08-06T22:03:46.175499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T14:59:19.528028Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20691","last_updated":"2025-08-28T11:50:22Z","snapshot_observed_at":"2026-08-06T13:33:55.665527Z","submitted_at":"2025-08-28T11:50:22Z","title":"MobileCLIP2: Improving Multi-Modal Reinforced Training","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-05T14:59:19.528028Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2508.20691"},"observation_digest":"sha256:043a6205dafc59cca6bb825e476b4a022e98ae4f52e0249191d6a206fde3d4c1","observation_id":"6823ab5d-aa99-4459-8da0-55367a9087ae","resolution":{"observed_at":"2026-08-05T14:59:19.528028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T12:22:33.659440Z","title":"Dat- acomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01644","last_updated":"2025-09-01T17:38:21Z","snapshot_observed_at":"2026-08-08T14:55:06.188124Z","submitted_at":"2025-09-01T17:38:21Z","title":"OpenVision 2: A Family of Generative Pretrained Visual Encoders for Multimodal Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T12:22:33.659440Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2509.01644"},"observation_digest":"sha256:13d04f1899bf57773807e93eb6d4e578b43bab04bf2777d974452565ad613f1f","observation_id":"98faa756-30e0-4b35-a4e4-807594854e0b","resolution":{"observed_at":"2026-08-05T12:22:33.659440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2510.04056","last_updated":"2026-05-12T17:38:57Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T06:34:30Z","title":"QuiLL: An LLM-Based Vulnerability Assessment Framework for the Wild","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-18T10:56:28.973065Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2510.04056"},"observation_digest":"sha256:edf436ca05b0bdb0469954825dbacbe224704a71af56d2d9a9c1abf86ccacd91","observation_id":"ba5b2e99-95be-4cbb-8b2c-e65ec2e8cb55","resolution":{"observed_at":"2026-05-18T11:01:17.173725Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2604.14198","last_updated":"2026-04-03T04:26:43Z","snapshot_observed_at":"2026-08-06T11:58:56.492844Z","submitted_at":"2026-04-03T04:26:43Z","title":"MixAtlas: Uncertainty-aware Data Mixture Optimization for Multimodal LLM Midtraining","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T20:07:29.544613Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2604.14198"},"observation_digest":"sha256:fd8b51c574757c12a1ba4aa941b1e4b6ae8d81190892136ef10f2a33974d94bf","observation_id":"36a23a66-588a-4566-86fd-137999944328","resolution":{"observed_at":"2026-05-13T20:08:12.623943Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2604.25154","last_updated":"2026-04-28T02:56:17Z","snapshot_observed_at":"2026-07-06T23:11:02.043135Z","submitted_at":"2026-04-28T02:56:17Z","title":"Prior-Aligned Data Cleaning for Tabular Foundation Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-07T16:47:31.905149Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2604.25154"},"observation_digest":"sha256:56e88132f7866ed29c8bab9d64e7ad890a2dbbb4567a906548eec82675da3539","observation_id":"354b2574-26db-433b-b021-4ab0150da4c1","resolution":{"observed_at":"2026-05-11T23:31:17.041197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.05416","last_updated":"2026-05-06T20:20:17Z","snapshot_observed_at":"2026-08-03T01:45:26.187986Z","submitted_at":"2026-05-06T20:20:17Z","title":"From Cradle to Cloud: A Life Cycle Review of AI's Environmental Footprint","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-08T15:43:50.422887Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.05416"},"observation_digest":"sha256:42a3cd0226892230ca5bfe3630d91891927effd698c5f0bc91e0164455e21b2d","observation_id":"44239207-3e69-40cd-b826-4e2a54fcd34f","resolution":{"observed_at":"2026-05-11T18:31:12.927636Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.09433","last_updated":"2026-05-10T09:13:40Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T09:13:40Z","title":"Offline Preference Optimization for Rectified Flow with Noise-Tracked Pairs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T02:10:27.595446Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.09433"},"observation_digest":"sha256:b35adb5eb6df0405a0cb57a59bc72ca082c7c696254b7878a8fc1cd98998a024","observation_id":"dd4dc2db-6d56-4ac3-be94-13ed77e75d64","resolution":{"observed_at":"2026-05-12T02:11:15.568450Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.20278","last_updated":"2026-05-24T12:22:09Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T04:39:28Z","title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T08:36:27.676888Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.20278"},"observation_digest":"sha256:d316799d3b5b13d583003d8f88d1d59459d93f14840d3cf3711e4aa61252cb02","observation_id":"fb96e050-6f1b-4a26-8252-11400aca9b15","resolution":{"observed_at":"2026-05-21T08:39:53.732465Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.20278","last_updated":"2026-05-24T12:22:09Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T04:39:28Z","title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T17:57:47.409741Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.20278"},"observation_digest":"sha256:757315aa7fe767ff8f2e9a674023717c1d38deed45575830ccc7c6d4381060de","observation_id":"cc80062b-c6f6-46c5-be8d-15b48b4c6989","resolution":{"observed_at":"2026-06-30T18:04:58.323685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.30341","last_updated":"2026-05-28T17:59:26Z","snapshot_observed_at":"2026-07-06T23:39:38.845840Z","submitted_at":"2026-05-28T17:59:26Z","title":"GPIC: A Giant Permissive Image Corpus for Visual Generation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T07:36:21.262064Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.30341"},"observation_digest":"sha256:8a015332779020e41502fa5de475c869741cabaa868cb35f36d0fef0a3dca25c","observation_id":"bffc5f5e-6bfb-4ee6-a4e5-d39898cd29a4","resolution":{"observed_at":"2026-06-29T07:43:14.139724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.06624","last_updated":"2026-06-08T15:12:03Z","snapshot_observed_at":"2026-08-02T09:46:22.331867Z","submitted_at":"2026-06-04T18:21:03Z","title":"Principles and Practice of Deep Representation Learning: or a Mathematical Theory of Memory","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T03:07:52.730713Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.06624"},"observation_digest":"sha256:4d998fba7727f655e6c0a113f9fcb0b7afecffa149adb24f3f312c42ff2e6aaf","observation_id":"7fab5c0d-92cc-4308-b2b2-1ece77cb957d","resolution":{"observed_at":"2026-07-02T11:46:55.266557Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.07865","last_updated":"2026-06-05T21:53:39Z","snapshot_observed_at":"2026-07-06T23:47:29.065892Z","submitted_at":"2026-06-05T21:53:39Z","title":"Instrumented data for causal scientific machine learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T22:24:28.643956Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.07865"},"observation_digest":"sha256:ca54264a96c73851cc27530a2825279470285c328b44099a4bd99f4a7222761a","observation_id":"8becee99-1461-4517-9165-db60f18cc0b9","resolution":{"observed_at":"2026-06-27T22:31:21.232799Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:795246144089b4aa0ead90ce2822844c7a48332ee470baf8fe49950dd457776e","observation_id":"02ffbe7a-5687-4cdd-a96d-6257e73f5924","resolution":{"observed_at":"2026-07-03T17:18:43.855348Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-08T09:42:45.507846Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-26T00:29:41.291832Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:72ab5a33ee322dbd12f7796354d0b561343811f15d6c4fa8d833ac9c311ba095","observation_id":"1ed58a5c-454a-49fe-9e34-bd224feed00c","resolution":{"observed_at":"2026-07-04T16:29:57.904493Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-08T09:42:45.507846Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T05:32:26.746776Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:2352f5fac4d89d8e69255492279313aa2f813170c40a2142bce9fb1610df60f5","observation_id":"60f42309-0369-402a-a0f7-a289360b21fd","resolution":{"observed_at":"2026-06-29T15:03:32.239700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.26199","last_updated":"2026-06-26T02:47:28Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T16:23:10Z","title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T01:52:12.291420Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.26199"},"observation_digest":"sha256:aae780540ee17b30327b9b181edbf5ba9d17ba6c6cf25ba03e2c0fb579c32de7","observation_id":"06c80a0a-6a21-4901-9a71-0ac0cf997d2b","resolution":{"observed_at":"2026-07-04T15:09:54.596740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.26199","last_updated":"2026-06-26T02:47:28Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T16:23:10Z","title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T04:46:56.552601Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.26199"},"observation_digest":"sha256:a0c880ea81d39267bae638b821b0ead2b0df6bb55236d81ae9d05a9b4917f681","observation_id":"ad1a47ec-70b8-44bb-9cfa-7ce7b22c1616","resolution":{"observed_at":"2026-06-29T19:13:53.521084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-07-12T01:07:20.766474Z","title":"Efros, Jenia Jitsev, Yair Carmon, Lud- wig Schmidt, and Vaishaal Shankar","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03624","last_updated":"2026-07-03T22:58:54Z","snapshot_observed_at":"2026-08-01T21:59:16.066897Z","submitted_at":"2026-07-03T22:58:54Z","title":"RADIO1D: Elastic Representations for Condensed Vision Modeling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T01:07:20.766474Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.03624"},"observation_digest":"sha256:5ae25fe4e0fb9cd62efc366457e3ebc0b1f044f34621fc95b8d4659e30efcac1","observation_id":"64f9987d-9b2e-495e-97c9-e375d3f97e63","resolution":{"observed_at":"2026-07-12T01:07:20.766474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv:2304.14108 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-06T12:36:13.363835Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:8f467f558af2fac4e940a677fd5d9955506e14e3fac72a2f64873afdddc3cd65","observation_id":"139999be-369b-4d71-8716-c21c3f3d0e95","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-01T01:15:07.149546Z","title":"2304.14108 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25886","last_updated":"2026-07-28T15:46:41Z","snapshot_observed_at":"2026-08-07T09:45:34.637759Z","submitted_at":"2026-07-28T15:46:41Z","title":"RSIBench-Data: Benchmarking Data-Centric Research for Recursive Self-Improvement","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T01:15:07.149546Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.25886"},"observation_digest":"sha256:e16b90aa7292199b833b44fb0db37f08ea2899639a62a2b46f0a03fe691ef15b","observation_id":"cf286f86-7e59-4e71-9d37-597deed39d0f","resolution":{"observed_at":"2026-08-01T01:15:07.149546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.14108/citation-record","integrity":"/paper/2304.14108/integrity","json":"/paper/2304.14108/citation-record.json","paper":"/paper/2304.14108"},"outbound":[],"paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2304.14108."}