{"as_of":"2026-08-18T22:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6d9fdb640c3c9509b44158de798739a6a86ecb25298b90256e8047463da25f9e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:10:19.293396Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T22:26:17.900939Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-08-16T17:13:28.893408Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:c9f571c3892ac5418839a01d8d8ad79ed1b5332c8a59c9c396cbf0e47731ed4f","observation_id":"e1a8c733-b4fb-417e-8c6c-509b6f930665","resolution":{"observed_at":"2026-05-24T09:04:15.092474Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2305.03726","last_updated":"2025-07-28T05:33:36Z","snapshot_observed_at":"2026-07-06T15:23:52.761222Z","submitted_at":"2023-05-05T17:59:46Z","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","version":2},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-05-15T02:43:47.775691Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2305.03726"},"observation_digest":"sha256:8b10b938836d9a4feb82a2c7c68d631aed056eeb00f509c41b46ccc4393f52b7","observation_id":"ce814b00-ab6d-48e6-81f0-0f66a3411612","resolution":{"observed_at":"2026-05-15T02:43:47.932640Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:f440800d9e94ad4f513deb7d8f611d73db65debeecac8e0ddfca77cb1182229b","observation_id":"83655539-7804-4fc7-a7dd-c87582ab9c46","resolution":{"observed_at":"2026-05-15T06:30:22.700766Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-14T01:52:01.163900Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2308.01390"},"observation_digest":"sha256:69823011c5a1f461a952cbc27d70d4fcac224be2a097c3910c6ff41fde54e0b7","observation_id":"00094086-cbec-421b-98d8-166b7721447f","resolution":{"observed_at":"2026-05-14T01:52:01.385739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2309.15112","last_updated":"2023-12-14T17:21:39Z","snapshot_observed_at":"2026-08-17T05:14:27.567959Z","submitted_at":"2023-09-26T17:58:20Z","title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","version":5},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-05-17T13:48:48.661566Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2309.15112"},"observation_digest":"sha256:b607de304be4fdd9f191db23042658e4593e72751ae3ad752fed2d385696aad4","observation_id":"c1907c8b-6166-4e65-8e0b-6357b6777362","resolution":{"observed_at":"2026-05-17T13:48:48.745647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2309.16671","last_updated":"2025-11-23T00:34:43Z","snapshot_observed_at":"2026-08-17T12:17:37.741844Z","submitted_at":"2023-09-28T17:59:56Z","title":"Demystifying CLIP Data","version":6},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-16T09:20:20.143143Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2309.16671"},"observation_digest":"sha256:a3609e755c926017dfb592c5183bd92709e631483ceec5cd07ec2b5ade3ad3d0","observation_id":"8b267e48-0291-4722-8ad5-92c0ee025391","resolution":{"observed_at":"2026-05-16T09:20:20.241274Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2404.14396","last_updated":"2025-03-02T07:53:44Z","snapshot_observed_at":"2026-08-12T19:12:43.246639Z","submitted_at":"2024-04-22T17:56:09Z","title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-15T22:48:36.010306Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2404.14396"},"observation_digest":"sha256:713ff928dda034d21ba8e45fb4002b16c3b59a15f2a42f0c85c19246367668c4","observation_id":"e5128dea-3a0d-4b63-83f8-af056b50585b","resolution":{"observed_at":"2026-05-15T22:48:36.123983Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2410.18715","last_updated":"2024-10-24T13:19:22Z","snapshot_observed_at":"2026-08-12T22:24:45.924222Z","submitted_at":"2024-10-24T13:19:22Z","title":"ChatSearch: a Dataset and a Generative Retrieval Model for General Conversational Image Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T18:29:18.466941Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2410.18715"},"observation_digest":"sha256:adc141a69791ea46d09142db54a28a29b454dd3ece7b829dba5a638f360b26de","observation_id":"78b15180-5f55-4b42-9b42-fd794e3c1335","resolution":{"observed_at":"2026-05-23T18:33:19.825940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-08-10T21:10:19.293396Z","title":"Multimodal C4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.05901","last_updated":"2025-01-13T02:34:19Z","snapshot_observed_at":"2026-08-14T04:25:59.263298Z","submitted_at":"2025-01-10T11:53:46Z","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T21:10:19.293396Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2501.05901"},"observation_digest":"sha256:587a948b958d5d6535e67fadafa3dbbf99136f1963c21028f0fde969188735db","observation_id":"0489dc65-fdfd-472b-9299-b388e7b33607","resolution":{"observed_at":"2026-08-10T21:10:19.293396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-08-08T12:31:52.057606Z","title":"ArXivabs/2304.06939 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.07527","last_updated":"2025-06-20T05:18:13Z","snapshot_observed_at":"2026-08-17T16:10:17.409163Z","submitted_at":"2025-02-11T13:08:03Z","title":"Nature Language Model: Deciphering the Language of Nature for Scientific Discovery","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T12:31:52.057606Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2502.07527"},"observation_digest":"sha256:092be83171a8d8f93aa0a3c5143a32220f555626663e0e3aabf4a894d5f54d6d","observation_id":"6ff885fa-82d4-455c-a507-07dc5c9787cd","resolution":{"observed_at":"2026-08-08T12:31:52.057606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-08-06T19:37:06.458067Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05240","last_updated":"2026-07-09T06:53:43Z","snapshot_observed_at":"2026-08-18T02:14:46.902275Z","submitted_at":"2025-07-07T17:49:41Z","title":"StreamVLN: Streaming Vision-and-Language Navigation via SlowFast Context Modeling","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:37:06.458067Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2507.05240"},"observation_digest":"sha256:f7ad1dd699cf0589304f4bab1037cfbd9dd894785fcec191291a88784fe04eff","observation_id":"f7ca8001-b5f3-4738-bb41-372f06b969e9","resolution":{"observed_at":"2026-08-06T19:37:06.458067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-08-06T00:59:54.830726Z","title":"Y.; Dodge, J.; Fang, A.; Yu, Y.; Schmidt, L.; Wang, W","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04088","last_updated":"2025-08-07T03:52:48Z","snapshot_observed_at":"2026-08-08T12:10:28.774282Z","submitted_at":"2025-08-06T05:10:29Z","title":"GM-PRM: A Generative Multimodal Process Reward Model for Multimodal Mathematical Reasoning","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T00:59:54.830726Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2508.04088"},"observation_digest":"sha256:db16d76d8c86b33f744bf4c51df359337b03208e5cd51cd46ac9e2745ef834a1","observation_id":"2b5a7845-eb60-4c4a-a109-8d51fbb49255","resolution":{"observed_at":"2026-08-06T00:59:54.830726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2603.27064","last_updated":"2026-04-14T20:08:27Z","snapshot_observed_at":"2026-08-18T14:08:51.713230Z","submitted_at":"2026-03-28T00:45:05Z","title":"ChartNet: A Million-Scale, High-Quality Multimodal Dataset for Robust Chart Understanding","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-14T22:39:06.113655Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2603.27064"},"observation_digest":"sha256:2604b0d13294a0fd1e80d56568f81c11a59e9b0d72f825d2adfe03fc87d99f08","observation_id":"922ea791-7568-45c2-a353-09ef6d0314ad","resolution":{"observed_at":"2026-05-14T22:39:32.927393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2605.12882","last_updated":"2026-05-13T01:54:42Z","snapshot_observed_at":"2026-08-11T13:11:09.870612Z","submitted_at":"2026-05-13T01:54:42Z","title":"CiteVQA: Benchmarking Evidence Attribution for Trustworthy Document Intelligence","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-14T20:37:36.144960Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2605.12882"},"observation_digest":"sha256:2b97ea6f4b01050b2679ef0a9d6241d5d2bc5e0ac6b1f1407e81af7ee90d650c","observation_id":"2a8080ea-7ead-49f4-951a-d2b9dba40426","resolution":{"observed_at":"2026-05-14T20:37:58.324569Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2606.07639","last_updated":"2026-06-01T09:07:15Z","snapshot_observed_at":"2026-08-17T19:24:43.936240Z","submitted_at":"2026-06-01T09:07:15Z","title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:31.310003Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2606.07639"},"observation_digest":"sha256:999ea6f6c2e844200700b30aff647671755f80769c925d93a51257eb99b3936a","observation_id":"6453e69f-deaf-40f8-811f-7358abe13d2d","resolution":{"observed_at":"2026-07-01T22:26:17.902355Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2304.06939/citation-record","integrity":"/paper/2304.06939/integrity","json":"/paper/2304.06939/citation-record.json","paper":"/paper/2304.06939"},"outbound":[],"paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T01:21:02.889451Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 15 inbound Pith citation observations for arXiv:2304.06939."}