{"as_of":"2026-08-18T12:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:87f1102174bd344b54de44d7392c121a281257571ea1783b3c377b9eadb95f87","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T12:57:12.054681Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:29:57.356586Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T13:39:50.707139Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-08-15T23:29:57.356586Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.04601","last_updated":"2025-05-07T17:48:35Z","snapshot_observed_at":"2026-08-16T12:58:27.250211Z","submitted_at":"2025-05-07T17:48:35Z","title":"OpenVision: A Fully-Open, Cost-Effective Family of Advanced Vision Encoders for Multimodal Learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T23:29:57.356586Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2505.04601"},"observation_digest":"sha256:1eb5acc43ebcbd9c9a40db50ce4f59f62328311da746e41980a5651457902115","observation_id":"ffcd5aec-4fd6-4615-9460-aa3aa72f2de1","resolution":{"observed_at":"2026-08-15T23:29:57.356586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-08-15T18:09:27.053527Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18915","last_updated":"2025-07-25T03:15:16Z","snapshot_observed_at":"2026-08-16T04:40:58.556320Z","submitted_at":"2025-07-25T03:15:16Z","title":"Mining Contextualized Visual Associations from Images for Creativity Understanding","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-15T18:09:27.053527Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2507.18915"},"observation_digest":"sha256:dc935cc10b945c8891ac93f6a33dcccebf7fbc1bcde9f24547c9f96697120edc","observation_id":"94e9f23b-b96e-4e8a-a73a-bc3f5e4dfeb2","resolution":{"observed_at":"2026-08-15T18:09:27.053527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-08-05T14:59:19.577024Z","title":"Scaling language-image pre-training via masking","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20691","last_updated":"2025-08-28T11:50:22Z","snapshot_observed_at":"2026-08-17T20:28:48.522116Z","submitted_at":"2025-08-28T11:50:22Z","title":"MobileCLIP2: Improving Multi-Modal Reinforced Training","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T14:59:19.577024Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2508.20691"},"observation_digest":"sha256:2f9e8c3812f4b00e94fdb0a1ae38a14fa319d480dff52addd7d1d09bad4ce523","observation_id":"f876f1b0-ae10-46cd-b18b-7130a3aece66","resolution":{"observed_at":"2026-08-05T14:59:19.577024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-08-05T12:22:35.606923Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.01644","last_updated":"2025-09-01T17:38:21Z","snapshot_observed_at":"2026-08-11T21:19:23.858904Z","submitted_at":"2025-09-01T17:38:21Z","title":"OpenVision 2: A Family of Generative Pretrained Visual Encoders for Multimodal Learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T12:22:35.606923Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2509.01644"},"observation_digest":"sha256:81e3b386379f98231cf67af2402d11cbfef154ce694a9c846780f311966688ea","observation_id":"bd5172db-a689-4111-ae86-087c771bf7ab","resolution":{"observed_at":"2026-08-05T12:22:35.606923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":"2411.16828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-07-04T13:39:50.707139Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":"f109aa76-2a13-46ea-bf1e-26a0697116ab","year":2024},"citing_paper":{"arxiv_id":"2605.00809","last_updated":"2026-06-09T13:08:44Z","snapshot_observed_at":"2026-08-16T17:50:54.118450Z","submitted_at":"2026-05-01T17:51:38Z","title":"Let ViT Speak: Generative Language-Image Pre-training","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-09T18:56:51.627714Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2605.00809"},"observation_digest":"sha256:1d7b85000fa64075d84ede147f0634b0e629f7b41b46c0f179281af01cbd8d30","observation_id":"7f7bbdb0-2b4d-4391-9dce-876277e74df8","resolution":{"observed_at":"2026-05-11T16:01:07.255186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":"2411.16828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-07-04T13:39:50.707139Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":"f109aa76-2a13-46ea-bf1e-26a0697116ab","year":2024},"citing_paper":{"arxiv_id":"2605.00809","last_updated":"2026-06-09T13:08:44Z","snapshot_observed_at":"2026-08-16T17:50:54.118450Z","submitted_at":"2026-05-01T17:51:38Z","title":"Let ViT Speak: Generative Language-Image Pre-training","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-01T07:35:07.825460Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2605.00809"},"observation_digest":"sha256:6da38800a296a7dcccc3903698007cc442d8701b789056960350a371d78edebe","observation_id":"e7a00b24-9bbb-4774-a8e9-96d0caa49fbf","resolution":{"observed_at":"2026-07-01T07:35:28.764723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":"2411.16828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-07-04T13:39:50.707139Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":"f109aa76-2a13-46ea-bf1e-26a0697116ab","year":2024},"citing_paper":{"arxiv_id":"2606.03713","last_updated":"2026-08-11T09:33:56Z","snapshot_observed_at":"2026-08-14T23:09:30.355546Z","submitted_at":"2026-06-02T14:34:48Z","title":"Investigating Adversarial Robustness of Multi-modal Large Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T11:11:34.152223Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2606.03713"},"observation_digest":"sha256:6100fe1b4987d728e707bf762aa16d10b5f14f13f88b995f1ae9f573bc4650f9","observation_id":"c35d4c3c-47fc-4f60-bc6e-2dff7871274c","resolution":{"observed_at":"2026-07-02T02:06:27.577528Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":"2411.16828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-07-04T13:39:50.707139Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":"f109aa76-2a13-46ea-bf1e-26a0697116ab","year":2024},"citing_paper":{"arxiv_id":"2606.03730","last_updated":"2026-08-11T09:26:07Z","snapshot_observed_at":"2026-08-15T00:55:31.878235Z","submitted_at":"2026-06-02T14:49:04Z","title":"Beyond False Stability: High-Noise Drift Gating for Test-Time Adversarial Defenses in Vision-Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T11:04:30.654255Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2606.03730"},"observation_digest":"sha256:8189790b9dc8d186d5891610b15f11877eb7032375693ad5873117976d1d551c","observation_id":"66670d96-e570-4654-bdfa-0efce984068c","resolution":{"observed_at":"2026-07-02T02:16:26.830996Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"cited_work":{"arxiv_id":"2411.16828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.16828","snapshot_observed_at":"2026-07-04T13:39:50.707139Z","title":"Clips: An enhanced clip framework for learning with synthetic captions","venue":null,"work_id":"f109aa76-2a13-46ea-bf1e-26a0697116ab","year":2024},"citing_paper":{"arxiv_id":"2606.26794","last_updated":"2026-06-25T09:27:54Z","snapshot_observed_at":"2026-08-16T14:44:55.956830Z","submitted_at":"2026-06-25T09:27:54Z","title":"ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-26T05:03:15.044146Z"},"links":{"cited_paper":"/paper/2411.16828","citing_paper":"/paper/2606.26794"},"observation_digest":"sha256:4dd32dcbed47c17db234643feb15cd46a1d93b4f65142fcc264153d2d4549c1f","observation_id":"46823c8a-a6ee-4e93-945b-adc5fa8d4ea2","resolution":{"observed_at":"2026-07-04T13:39:50.708849Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.16828/citation-record","integrity":"/paper/2411.16828/integrity","json":"/paper/2411.16828/citation-record.json","paper":"/paper/2411.16828"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T12:57:11.662098Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.662098Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:f92747d7c48aafc4c0702adc14321ad74a7f24072c018c421cde27f603fe6c0d","observation_id":"87a9bce5-8af8-4d75-aece-df6652e9943c","resolution":{"observed_at":"2026-08-12T12:57:11.662098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.386012Z","title":"nocaps: novel object caption- ing at scale","venue":null,"work_id":"d361d8ff-76f3-4ba4-99e1-f90fb33bde90","year":2019},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.668803Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:602f9b3d16423abe647601075e32e63fb3049d244581436aaa7f4bdf15e19b08","observation_id":"5d27400c-8ca4-4ae7-9221-91b6b7517cc0","resolution":{"observed_at":"2026-08-12T12:57:13.393647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-12T12:57:11.674832Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.674832Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:ef207df44834ed243cfef1c7bbd4e5f8b408ebdc54910a3de22a8efe90b44a1e","observation_id":"9c7aa08c-03a6-4128-be0d-7933828bdbe0","resolution":{"observed_at":"2026-08-12T12:57:11.674832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02746","last_updated":"2025-02-19T05:59:59Z","snapshot_observed_at":"2026-08-16T13:12:51.543633Z","submitted_at":"2024-10-03T17:56:09Z","title":"Contrastive Localized Language-Image Pre-Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02746","snapshot_observed_at":"2026-08-12T12:57:11.682472Z","title":"Contrastive localized language- image pre-training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.682472Z"},"links":{"cited_paper":"/paper/2410.02746","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:604b04ca8d5956c57e1d64865ac786c45b3f9e1ecf3bd880cf30f502f2616c6a","observation_id":"d0a64e86-c10a-46ae-b785-ac4b59bec140","resolution":{"observed_at":"2026-08-12T12:57:11.682472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04692","last_updated":"2024-03-17T16:59:25Z","snapshot_observed_at":"2026-08-16T14:11:52.899972Z","submitted_at":"2024-03-07T17:41:37Z","title":"PixArt-\\Sigma: Weak-to-Strong Training of Diffusion Transformer for 4K Text-to-Image Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04692","snapshot_observed_at":"2026-08-12T12:57:11.689555Z","title":"Pixart- \\sigma: Weak-to-strong training of diffusion transformer for 4k text-to-image generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.689555Z"},"links":{"cited_paper":"/paper/2403.04692","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:e5483655291a36ff8b4464607795873eeec1414c12129247677f1a3a18cd07bf","observation_id":"d865697b-4c7d-4f3d-8ded-4c99fe4736f3","resolution":{"observed_at":"2026-08-12T12:57:11.689555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-14T06:42:43.375489Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12793","snapshot_observed_at":"2026-08-12T12:57:11.695395Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.695395Z"},"links":{"cited_paper":"/paper/2311.12793","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:752e3685140dd8fdd235d92684e2bbb7df67571d958f8601ad2b1d01ea32e186","observation_id":"2fa93cab-c879-401a-93bf-f6af44de8b9a","resolution":{"observed_at":"2026-08-12T12:57:11.695395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.06794","last_updated":"2023-06-05T17:55:12Z","snapshot_observed_at":"2026-08-16T15:49:09.216982Z","submitted_at":"2022-09-14T17:24:07Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.06794","snapshot_observed_at":"2026-08-12T12:57:11.701555Z","title":"Pali: A jointly- scaled multilingual language-image model","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.701555Z"},"links":{"cited_paper":"/paper/2209.06794","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:20848a685f856a8090848d02d722d3046c45288ba9559bea36e88c301a3e0819","observation_id":"a6c63e58-115e-4a5c-9fcf-14d6cd4b683c","resolution":{"observed_at":"2026-08-12T12:57:11.701555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.710436Z","title":"Uniter: Universal image-text representation learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.710436Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:fb3c83b7b6a5b1437a1b79d0b1ba1fc4abe3603a30084c6ae0f7de6548a8a6ae","observation_id":"31d3cbf0-4c91-43ab-b5ec-87d41022b000","resolution":{"observed_at":"2026-08-12T12:57:11.710436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.721096Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.721096Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:3a78a6ad6ef18eecbe3590f51c4afe57b027bfaec0fc7e171c829af0adbc58b6","observation_id":"856507f4-b276-45cd-b87d-591fca256d3d","resolution":{"observed_at":"2026-08-12T12:57:11.721096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03766","last_updated":"2024-02-06T07:16:36Z","snapshot_observed_at":"2026-08-09T19:28:34.281681Z","submitted_at":"2024-02-06T07:16:36Z","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03766","snapshot_observed_at":"2026-08-12T12:57:11.727721Z","title":"Mobilevlm v2: Faster and stronger baseline for vision language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.727721Z"},"links":{"cited_paper":"/paper/2402.03766","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:1101ebb322ec86b6e3729cf6f0db37c70a0d0b1c097482e6cfda45773fd776e3","observation_id":"19eac34b-3930-450a-a17e-cc60b3b72e55","resolution":{"observed_at":"2026-08-12T12:57:11.727721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.734320Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.734320Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:ca172e9562ce9dfcf26bebe8de2ade7065f0cce597e2cbaa4df44ff90fc60f6d","observation_id":"f8190cd5-b06b-4614-854d-11611a82aa61","resolution":{"observed_at":"2026-08-12T12:57:11.734320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T12:57:11.743434Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.743434Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:826f11596df9f942c35146077b702b7f4d2d992cff38ffbd665dc4ca546065bf","observation_id":"f2ebdab8-4d15-4188-a959-e3da8492db67","resolution":{"observed_at":"2026-08-12T12:57:11.743434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-12T12:57:11.751560Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.751560Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:d319a14c457fe1c9c3105c1f1f7d0cc7007aa8d5aa68d4fd4e2d5ca0c24b079d","observation_id":"b9f5d061-4bfd-406c-acde-9f0c709a33cf","resolution":{"observed_at":"2026-08-12T12:57:11.751560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.309899Z","title":"Improving clip training with language rewrites","venue":null,"work_id":"d0d99989-9e12-4b00-988e-b688757be905","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.758151Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:fd344e0de2022511ab55cd28eab065a2c64f5f6077a3bdac9aeeab5bd0d73bbf","observation_id":"82a20a60-25b5-4c7f-a116-bb3244b52271","resolution":{"observed_at":"2026-08-12T12:57:13.320690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17425","last_updated":"2023-11-06T02:47:51Z","snapshot_observed_at":"2026-08-16T14:56:25.720519Z","submitted_at":"2023-09-29T17:37:29Z","title":"Data Filtering Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17425","snapshot_observed_at":"2026-08-12T12:57:11.763989Z","title":"Data fil- tering networks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.763989Z"},"links":{"cited_paper":"/paper/2309.17425","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:b1dfa05f6cbd097fc756e007ff134ce48fba3dd404323f8a367220ce7605bb7b","observation_id":"23714bab-d68e-4368-b4ad-f8eb6dbd2213","resolution":{"observed_at":"2026-08-12T12:57:11.763989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-12T12:57:11.771942Z","title":"Mme: A comprehensive evaluation bench- mark for multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.771942Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:4ca3c010da4d727a1a905622726e2ab96555ad7a446e9401aa546792c9ae0baa","observation_id":"d164bd10-f6a1-4f7e-a598-75097867be19","resolution":{"observed_at":"2026-08-12T12:57:11.771942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.265208Z","title":"Dat- acomp: In search of the next generation of multimodal datasets","venue":null,"work_id":"1c68ca04-030a-4dd9-a436-e61ecdada45e","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.780479Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:bc1229ed0c22f262a0bfca8bae9e1b036f6a3d28dfbee965feff3433f32545ea","observation_id":"675b36ef-7343-423b-a2b9-774c7b581dac","resolution":{"observed_at":"2026-08-12T12:57:13.277650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01832","last_updated":"2024-07-18T10:21:29Z","snapshot_observed_at":"2026-08-16T14:22:02.043874Z","submitted_at":"2024-02-02T18:59:58Z","title":"SynthCLIP: Are We Ready for a Fully Synthetic CLIP Training?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01832","snapshot_observed_at":"2026-08-12T12:57:11.786071Z","title":"Synthclip: Are we ready for a fully synthetic clip training? arXiv preprint arXiv:2402.01832, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.786071Z"},"links":{"cited_paper":"/paper/2402.01832","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:bb56978e68fc4c331999f3b5caf203513e7ea7e87426b9482572b7c121ab4d82","observation_id":"c0e496eb-6e35-46fc-8f0d-99aa2501f4af","resolution":{"observed_at":"2026-08-12T12:57:11.786071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.234739Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":"5d760984-eb0d-445b-bbf9-c7bb6d3f8b60","year":2019},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.807564Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:8d39fbe2bfc146ad156083018afbeb20c7fdd0b3779b1890b3cab9e60e926ad3","observation_id":"d0e9be35-437e-48ee-be88-39a63aab6713","resolution":{"observed_at":"2026-08-12T12:57:13.247902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.210025Z","title":"Open- clip","venue":null,"work_id":"168e7aef-d46e-472d-8b1a-d6b5721ba239","year":2021},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.817392Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:26f48e43fae0ad38155e2e5dc6af1278b974adcc238feab0cc5721c17a194393","observation_id":"7e1d1dc0-b2cb-4289-961c-7305b5a4e58e","resolution":{"observed_at":"2026-08-12T12:57:13.217337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.823008Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.823008Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:10ec7f18ab1741ceeac9eadb231e682e435de608f4a18512ce4611d83d48b3fe","observation_id":"4e663a27-f256-4864-8de8-a8110ad34162","resolution":{"observed_at":"2026-08-12T12:57:11.823008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.829469Z","title":"Vilt: Vision- and-language transformer without convolution or region su- pervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.829469Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:0eed52b2f9228bd05d09b653a965705cbadad6a4b4965e099bb1a24bc369c526","observation_id":"e77d857e-3b87-4a6d-9e35-25bcd6e60d66","resolution":{"observed_at":"2026-08-12T12:57:11.829469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.157450Z","title":"Veclip: Improving clip training via visual-enriched captions","venue":null,"work_id":"f6447b4a-7ac4-4d40-a564-9a122f8bcdc2","year":2025},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.835490Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:5ff2da5cf861b6e8e397998571d1cb40c0a969bcbee518e1e155433950d17fd7","observation_id":"2fafa437-3385-432e-ad67-33adf7348b28","resolution":{"observed_at":"2026-08-12T12:57:13.164438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00740","last_updated":"2025-03-29T12:57:07Z","snapshot_observed_at":"2026-08-16T13:56:46.712499Z","submitted_at":"2024-04-30T01:19:18Z","title":"Modeling Caption Diversity in Contrastive Vision-Language Pretraining","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00740","snapshot_observed_at":"2026-08-12T12:57:11.841835Z","title":"Modeling caption diversity in contrastive vision-language pretraining","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.841835Z"},"links":{"cited_paper":"/paper/2405.00740","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:a6ad1a512e985a83d20daf465e83d3b0b3579a3796daa9c8fcc28f9c03c5e472","observation_id":"9299736a-f2b8-4ea5-9d26-fa4dbc5a0f44","resolution":{"observed_at":"2026-08-12T12:57:11.841835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.849373Z","title":"Align before fuse: Vision and language representation learn- ing with momentum distillation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.849373Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:2eb9aa428f66313e0d500149350da4e9944c91ccb11df65d63338e3772fe2638","observation_id":"94c672f7-5eae-4a63-a2d5-3b53ef73108d","resolution":{"observed_at":"2026-08-12T12:57:11.849373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.857195Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.857195Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:739405ce13e6b19aa8958b2b29e9c7c04e7007902b6fb1d6d21a9a056fc5faf3","observation_id":"c46fb3ea-2ed2-4ca7-a8c7-00abf378e968","resolution":{"observed_at":"2026-08-12T12:57:11.857195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.863570Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.863570Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:994454250f45dfc220e11936e6c09005055a989ffe0e5b2e6429b45e352677f3","observation_id":"fed288c3-f5a6-467a-aa17-970573b7dd0f","resolution":{"observed_at":"2026-08-12T12:57:11.863570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03557","last_updated":"2019-08-09T17:57:13Z","snapshot_observed_at":"2026-08-07T01:39:40.489262Z","submitted_at":"2019-08-09T17:57:13Z","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03557","snapshot_observed_at":"2026-08-12T12:57:11.870361Z","title":"Visualbert: A simple and perfor- mant baseline for vision and language","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.870361Z"},"links":{"cited_paper":"/paper/1908.03557","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:3752a631603b9db9120abded5f37dca95a374e39ee6054bd688bd4cf23e6817e","observation_id":"d1171e4b-2022-4e26-9bcf-245de887639f","resolution":{"observed_at":"2026-08-12T12:57:11.870361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15658","last_updated":"2023-06-27T17:51:06Z","snapshot_observed_at":"2026-08-16T15:20:39.273734Z","submitted_at":"2023-06-27T17:51:06Z","title":"CLIPA-v2: Scaling CLIP Training with 81.1% Zero-shot ImageNet Accuracy within a \\$10,000 Budget; An Extra \\$4,000 Unlocks 81.8% Accuracy","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15658","snapshot_observed_at":"2026-08-12T12:57:11.877950Z","title":"Clipa-v2: Scaling clip training with 81.1% zero-shot imagenet accuracy within a $10,000 budget; an extra $4,000 unlocks 81.8% accuracy","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.877950Z"},"links":{"cited_paper":"/paper/2306.15658","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:29b2eedb64e772d4e02a022cde67caaf65ff9b0a7f0b8d0f15989d4f59615c3b","observation_id":"4886e482-f71b-40e5-baa4-cff250d1c4af","resolution":{"observed_at":"2026-08-12T12:57:11.877950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08478","last_updated":"2024-06-18T11:47:26Z","snapshot_observed_at":"2026-08-16T13:43:37.667967Z","submitted_at":"2024-06-12T17:59:07Z","title":"What If We Recaption Billions of Web Images with LLaMA-3?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08478","snapshot_observed_at":"2026-08-12T12:57:11.884915Z","title":"What if we recaption billions of web images with llama-3? arXiv preprint arXiv:2406.08478,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.884915Z"},"links":{"cited_paper":"/paper/2406.08478","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:20564a46961c0f060442588901238482e37095a3986379d12c84f157584d3d92","observation_id":"40b79ca7-7d68-4d30-ae59-f2fc638f8802","resolution":{"observed_at":"2026-08-12T12:57:11.884915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.089081Z","title":"An inverse scal- ing law for clip training","venue":null,"work_id":"8fa5a078-2a77-4135-94f7-83bb2d7ad152","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.890416Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:2f200db0f3f35896dce762c64c7fe39c866026d74c8f9792aa695556b8ba7974","observation_id":"118f6ee0-5eb3-442b-834f-a3b6a43de51f","resolution":{"observed_at":"2026-08-12T12:57:13.096232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.895324Z","title":"Evaluating object hallucination in large vision-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.895324Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:8483db724f6d1d86fd1a551a941d49c3fe5a2972eb9cec19fe567f13eec1cd5b","observation_id":"3d6e3517-fe59-4b9f-8675-3d1cd0181484","resolution":{"observed_at":"2026-08-12T12:57:11.895324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15947","snapshot_observed_at":"2026-08-12T12:57:11.900633Z","title":"Moe-llava: Mixture of experts for large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.900633Z"},"links":{"cited_paper":"/paper/2401.15947","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:0b3961d1ab386fccb6f77188b5924f779767c00f67de2c02ca66468a7666b745","observation_id":"2e13e9a0-7ba8-4341-8a23-3029afa06591","resolution":{"observed_at":"2026-08-12T12:57:11.900633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.912062Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.912062Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:b14a8ad649120be23a08de6bced021a55601275be04f33b0490837f726e226b8","observation_id":"b20fcd7b-2378-4c47-be2c-be2e2a52e369","resolution":{"observed_at":"2026-08-12T12:57:11.912062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:13.020466Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"1ce99f69-40b3-47ea-a39b-b1c9dfcfa91f","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.917940Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:a0308c1bc0953e4d17b717002aef3418b231cd0acd8f26177e4dfc6852f0bfe0","observation_id":"4c7bcbe9-72e3-4cc9-a67d-2f9137696731","resolution":{"observed_at":"2026-08-12T12:57:13.027445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.923586Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.923586Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:e5497e75bf4ce11543a199894a488228bba1bc4982ef9d3286e6177397b49853","observation_id":"974a6fed-75b7-42fe-a859-4a80f5947e52","resolution":{"observed_at":"2026-08-12T12:57:11.923586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18765","last_updated":"2024-03-13T08:47:32Z","snapshot_observed_at":"2026-08-18T02:01:07.410240Z","submitted_at":"2023-11-30T18:05:52Z","title":"MLLMs-Augmented Visual-Language Representation Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18765","snapshot_observed_at":"2026-08-12T12:57:11.929501Z","title":"Mllms- augmented visual-language representation learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.929501Z"},"links":{"cited_paper":"/paper/2311.18765","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:43c1cb77b65f8cd06bde5c8a34d2f050dfb9c03955dfb5ca6896d64ec713643a","observation_id":"2def70fc-4104-4b95-9f14-89fd72aaa89d","resolution":{"observed_at":"2026-08-12T12:57:11.929501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.937991Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.937991Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:6d3557f6568ad749ee907ccccf72895399fbd44097e7a3bc6d0406fd95ca62db","observation_id":"79aad07b-cc57-4341-828f-90a2f2aff7a7","resolution":{"observed_at":"2026-08-12T12:57:11.937991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.954664Z","title":"ChartQA: A benchmark for question answer- ing about charts with visual and logical reasoning","venue":null,"work_id":"d52a2757-5094-4b58-bda2-21a2ef0fb585","year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.942742Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:9907e8d6a03b1c91afc659321c3412b7cd63d2f9ade2f2f623b16f94a612f089","observation_id":"b816f311-eb6c-47b7-8a25-e3285280ef85","resolution":{"observed_at":"2026-08-12T12:57:12.961870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.933611Z","title":"Introducing meta llama 3: The most capable openly available llm to date","venue":null,"work_id":"67703cc7-146e-4edc-8405-ab1a58e741fa","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.950195Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:d2408c620fe5dc3adb3e71c20dde6f0f8f35140f0a68331b83033d5385e0003c","observation_id":"1f25dc0d-99d3-4021-8a8e-2ab38f1d389f","resolution":{"observed_at":"2026-08-12T12:57:12.940951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.908771Z","title":"Improving multimodal datasets with image captioning","venue":null,"work_id":"00d061b2-44c4-4e9a-bcff-0a084fd25e3d","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.957998Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:eb49a74c0b24c3ca880804a9a59b20ed5809f21971b913f0965da50e5725e59b","observation_id":"95029b9b-1135-47f5-9d57-a8d123894450","resolution":{"observed_at":"2026-08-12T12:57:12.918110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-08-14T18:53:38.574749Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-12T12:57:11.963589Z","title":"Repre- sentation learning with contrastive predictive coding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.963589Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:50bfffb04b96fcbd71fd5d0d00beed7c3690edcc39548e2f98e35f4558a30cf9","observation_id":"864f0c83-36d8-4123-8902-435d856882bb","resolution":{"observed_at":"2026-08-12T12:57:11.963589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.881107Z","title":"Introducing chatgpt","venue":null,"work_id":"ded8c71c-b17f-43b8-b467-60adaba4d6e7","year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.969254Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:5884b5c8109e523f4c8edbb7db60239a1cce9dcf104c5c542824d15dce98818b","observation_id":"9d418dd9-57d1-4748-8765-a003e296d0f6","resolution":{"observed_at":"2026-08-12T12:57:12.889022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:11.974372Z","title":"Flickr30k entities: Collecting region-to-phrase corre- spondences for richer image-to-sentence models","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.974372Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:23147796f4812aefbd59ab9ec9facaa28e13e470dbd2d0bcd1177a7b9a31f43a","observation_id":"7bc454e2-6576-4262-a68c-d1ddbcdcf4b4","resolution":{"observed_at":"2026-08-12T12:57:11.974372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.842243Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"8097653c-c64f-43e8-a54b-7c8e0665a949","year":2021},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.979462Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:f4d86e41129cc27501ae90d4f6c0a1a30bdca1af4bf49ea91e14974d0c0875ea","observation_id":"cdf99a94-d00c-45e2-846b-2441208eb8da","resolution":{"observed_at":"2026-08-12T12:57:12.850186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.822210Z","title":"Laion-5b: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":"3404d460-d92a-495c-b0f0-020622624db6","year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.987926Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:8a2d5ebdd0f8735e0ebfb68cf00f724b438d6f0af4caed18b2e85b8e5bf808e2","observation_id":"d2d1b4a7-cc30-46b9-b26b-46a15f778456","resolution":{"observed_at":"2026-08-12T12:57:12.828502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.797622Z","title":"Towards vqa models that can read","venue":null,"work_id":"0ce9f49a-72f5-4720-93c2-37ea7fe5d56c","year":2019},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:11.995298Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:8831f7513787bb6e2db34183309a3472e5b714cc61c5d206b15064b897cca3df","observation_id":"717c274e-ca7c-44aa-9a32-90881f0de1df","resolution":{"observed_at":"2026-08-12T12:57:12.805935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.769777Z","title":"Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","venue":null,"work_id":"d355487a-7cae-4a63-a30c-9f9052c1816f","year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.003199Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:d61a2401c8ca90b0d1a48f3f7a7753815aae1e05cc2d045c3646500a06e9bb06","observation_id":"35637afb-a9e7-457d-99da-76ec8784bc20","resolution":{"observed_at":"2026-08-12T12:57:12.778594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.07952","last_updated":"2024-03-17T06:49:19Z","snapshot_observed_at":"2026-08-17T14:12:03.932742Z","submitted_at":"2023-06-13T17:51:18Z","title":"MOFI: Learning Image Representations from Noisy Entity Annotated Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.07952","snapshot_observed_at":"2026-08-12T12:57:12.015082Z","title":"Mofi: Learning image represen- tations from noisy entity annotated images","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.015082Z"},"links":{"cited_paper":"/paper/2306.07952","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:7a9a5f337af4c921f3134ff3e216f92f88962fe93cda140e8174a048671dfdd6","observation_id":"d8bc428a-7b52-45df-a683-b0a6d8981a14","resolution":{"observed_at":"2026-08-12T12:57:12.015082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.021017Z","title":"Alip: Adaptive language-image pre-training with synthetic cap- tion","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.021017Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:72cfc43271f45da32858a5aa50a1b1af0a4ea9351e8153029e0907ccd1b871f6","observation_id":"b9f44da5-3e99-4cbb-80f9-d7d23d07b23e","resolution":{"observed_at":"2026-08-12T12:57:12.021017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.07783","last_updated":"2021-11-09T17:15:38Z","snapshot_observed_at":"2026-08-17T06:55:10.468196Z","submitted_at":"2021-11-09T17:15:38Z","title":"FILIP: Fine-grained Interactive Language-Image Pre-Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.07783","snapshot_observed_at":"2026-08-12T12:57:12.027307Z","title":"Filip: Fine-grained interactive language-image pre-training","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.027307Z"},"links":{"cited_paper":"/paper/2111.07783","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:aaaf870854cd66b67f6a5058fbd2456ac7d29d5cabdaaaad757c090853f9a8a0","observation_id":"dbd676ce-f04c-43ee-944b-1e300a7a59b0","resolution":{"observed_at":"2026-08-12T12:57:12.027307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01917","last_updated":"2022-06-14T00:48:04Z","snapshot_observed_at":"2026-08-13T15:12:42.567441Z","submitted_at":"2022-05-04T07:01:14Z","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01917","snapshot_observed_at":"2026-08-12T12:57:12.033282Z","title":"Coca: Contrastive captioners are image-text foundation models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.033282Z"},"links":{"cited_paper":"/paper/2205.01917","citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:f4cfc80b15ca476fa9a251eb095cd218b19325b2ce2baed9d57e09348fd2b980","observation_id":"f5cf8247-e0e0-49ea-a7ce-7f2cacd0fc0f","resolution":{"observed_at":"2026-08-12T12:57:12.033282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.738246Z","title":"Mmmu: A massive multi-discipline multimodal understand- ing and reasoning benchmark for expert agi","venue":null,"work_id":"f11b5af9-296c-46e7-92d9-681529052b3a","year":2024},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.040523Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:9ec05d46cd3b2f5236dafc99baa4a2c1587740eab921779525fcef8dffc10aa7","observation_id":"86a5d4ed-52e1-4694-9baa-893666e4f0ce","resolution":{"observed_at":"2026-08-12T12:57:12.744561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.717267Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"818f72fb-5aa2-43b6-ba39-601b7b1f3732","year":2023},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.048353Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:2390a7a8510e87defef050fc09337cec820697497a41ddc6462caddd75e2f43b","observation_id":"b7a435bc-b58a-4d30-bb5b-a618fd2e5103","resolution":{"observed_at":"2026-08-12T12:57:12.724421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:57:12.671471Z","title":"Dreamlip: Language- image pre-training with long captions","venue":null,"work_id":"88e9d2c1-3fbf-46e1-a43f-1815a89d0719","year":2025},"citing_paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T12:57:12.054681Z"},"links":{"citing_paper":"/paper/2411.16828"},"observation_digest":"sha256:81ea15d8bf081adef908b8ca760cfd03c737d50d41d3f5a0afb4300cb6a8a82a","observation_id":"0d5b9789-3e86-4abd-b014-b6b57742864a","resolution":{"observed_at":"2026-08-12T12:57:12.695816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.16828","last_updated":"2024-11-25T18:49:02Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T00:23:19.834960Z","submitted_at":"2024-11-25T18:49:02Z","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":0,"verified_fuzzy":19},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 9 inbound Pith citation observations for arXiv:2411.16828."}