{"as_of":"2026-08-16T08:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2f123d96d64ba37246b3cc3b7474dcb7da56bf6f1b30e123ccc482b54b0a4fdb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":50,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:18:15.201350Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T20:16:29.588718Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2403.18814","last_updated":"2024-03-27T17:59:04Z","snapshot_observed_at":"2026-07-31T05:41:28.385099Z","submitted_at":"2024-03-27T17:59:04Z","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T07:44:47.355960Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2403.18814"},"observation_digest":"sha256:30ad4bf383948b7fe35be9675bde1d03c283f1f818f7d92ae032d2893748bf70","observation_id":"70162ff3-a8ce-43d7-86d6-46e45d12b224","resolution":{"observed_at":"2026-05-17T07:44:47.453596Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2404.14396","last_updated":"2025-03-02T07:53:44Z","snapshot_observed_at":"2026-08-12T19:12:43.246639Z","submitted_at":"2024-04-22T17:56:09Z","title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T22:48:36.010306Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2404.14396"},"observation_digest":"sha256:e38a1336b4ac63fd31f40a35d44c6772e66473c0a7817bed4c0b01f9b34eff77","observation_id":"5892f0c8-53f8-4454-b1e4-d0210078372b","resolution":{"observed_at":"2026-05-15T22:48:36.390334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2405.08748","last_updated":"2024-05-14T16:33:25Z","snapshot_observed_at":"2026-07-06T18:14:16.386835Z","submitted_at":"2024-05-14T16:33:25Z","title":"Hunyuan-DiT: A Powerful Multi-Resolution Diffusion Transformer with Fine-Grained Chinese Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T14:58:37.383749Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2405.08748"},"observation_digest":"sha256:36deee1f121c1555dcc4394d5b5e912aacc2b4aa53365aba981b49a05a0132d8","observation_id":"239459fe-7563-48fd-a297-eca4d5de87d5","resolution":{"observed_at":"2026-05-16T14:58:37.478355Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2406.06525","last_updated":"2024-06-10T17:59:52Z","snapshot_observed_at":"2026-08-13T22:05:34.844117Z","submitted_at":"2024-06-10T17:59:52Z","title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T22:09:16.622717Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2406.06525"},"observation_digest":"sha256:d4ce2bc303474d268b8ef1ab523e0a4ce7c7e6875241dbd8e43d1e40b64064d1","observation_id":"9ff8c8a9-dfd0-4fc9-850a-e8352882a5e2","resolution":{"observed_at":"2026-05-11T22:09:16.881241Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2407.12772","last_updated":"2025-05-05T04:48:45Z","snapshot_observed_at":"2026-08-14T07:51:18.307451Z","submitted_at":"2024-07-17T17:51:53Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T05:19:22.423762Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2407.12772"},"observation_digest":"sha256:3e6d6035b45c244dabffd5f4484c2b7e84a2c36ca9291e5ec19e724a047ea302","observation_id":"46622577-559c-4ea9-afda-31666510f829","resolution":{"observed_at":"2026-05-17T05:19:22.459326Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2410.13848","last_updated":"2024-10-17T17:58:37Z","snapshot_observed_at":"2026-08-03T03:16:45.693966Z","submitted_at":"2024-10-17T17:58:37Z","title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T22:09:16.001309Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2410.13848"},"observation_digest":"sha256:3fc190fc4481b3fb9070f68109fd3ad86600b99bc4e0faf1dfdd4cba6d22a967","observation_id":"6412ae67-e7cb-450c-b736-4f27b31d55a5","resolution":{"observed_at":"2026-05-15T22:09:16.479085Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-12T20:10:24.255889Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09955","last_updated":"2024-11-21T05:28:10Z","snapshot_observed_at":"2026-08-16T03:54:44.637613Z","submitted_at":"2024-11-15T05:18:15Z","title":"Instruction-Guided Editing Controls for Images and Multimedia: A Survey in LLM era","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T20:10:24.255889Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2411.09955"},"observation_digest":"sha256:264fed1a7edb432d28cee459c5c4b783fd8ffc38522f3698ea82030da7042fde","observation_id":"6a992442-55ca-4ece-8619-4e6d2a23cb7b","resolution":{"observed_at":"2026-08-12T20:10:24.255889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-12T12:37:38.366549Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.17762","last_updated":"2025-07-28T09:54:49Z","snapshot_observed_at":"2026-08-13T06:27:52.789046Z","submitted_at":"2024-11-26T03:33:52Z","title":"MUSE-VL: Modeling Unified VLM through Semantic Discrete Encoding","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T12:37:38.366549Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2411.17762"},"observation_digest":"sha256:4bfc2d18acbce1c0543ff57b0f1bd7119f2a8d91696636998b77d8c34720e8b7","observation_id":"050a9970-6069-4910-b918-cb01401a1834","resolution":{"observed_at":"2026-08-12T12:37:38.366549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-12T11:10:54.624006Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.18499","last_updated":"2025-03-30T07:22:46Z","snapshot_observed_at":"2026-08-16T00:08:27.847640Z","submitted_at":"2024-11-27T16:39:04Z","title":"OpenING: A Comprehensive Benchmark for Judging Open-ended Interleaved Image-Text Generation","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T11:10:54.624006Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2411.18499"},"observation_digest":"sha256:851e39aa8af4f47f7b0742a538722b75fd7de9325e3d65bc6a7fd44043a0d036","observation_id":"f5df30e3-daed-4e0e-b44a-332eaf4051c5","resolution":{"observed_at":"2026-08-12T11:10:54.624006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-12T00:56:59.580658Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01824","last_updated":"2025-08-27T12:26:39Z","snapshot_observed_at":"2026-08-15T01:47:14.531478Z","submitted_at":"2024-12-02T18:59:26Z","title":"X-Prompt: Towards Universal In-Context Image Generation in Auto-Regressive Vision Language Foundation Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T00:56:59.580658Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.01824"},"observation_digest":"sha256:05c40f4eddfb056e9396059c2cc8152f053052b31f9387a0ed6b81952bd49c0d","observation_id":"2a352133-5948-4b17-8b7c-4bd3dd3493a4","resolution":{"observed_at":"2026-08-12T00:56:59.580658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T21:30:08.258608Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.04432","last_updated":"2024-12-05T18:53:04Z","snapshot_observed_at":"2026-08-14T13:59:34.648701Z","submitted_at":"2024-12-05T18:53:04Z","title":"Divot: Diffusion Powers Video Tokenizer for Comprehension and Generation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T21:30:08.258608Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.04432"},"observation_digest":"sha256:ef28f2ddf18582c5c55bd066906697b74cb946fe31c04b25165a2044f48617e3","observation_id":"340650dd-c441-4c0e-abe9-d8a31787b639","resolution":{"observed_at":"2026-08-11T21:30:08.258608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T21:30:00.136000Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.04447","last_updated":"2025-04-11T07:10:02Z","snapshot_observed_at":"2026-08-14T15:38:54.058153Z","submitted_at":"2024-12-05T18:57:23Z","title":"EgoPlan-Bench2: A Benchmark for Multimodal Large Language Model Planning in Real-World Scenarios","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T21:30:00.136000Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.04447"},"observation_digest":"sha256:476348081c686e08c358c0e05eceb6e4245f2e72ef298c8791cf9c3c70a02c65","observation_id":"3179b8c2-6059-45e9-9987-8cb2b594a1a3","resolution":{"observed_at":"2026-08-11T21:30:00.136000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T19:31:43.789931Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.06660","last_updated":"2024-12-09T16:59:35Z","snapshot_observed_at":"2026-08-13T09:14:45.905037Z","submitted_at":"2024-12-09T16:59:35Z","title":"MuMu-LLaMA: Multi-modal Music Understanding and Generation via Large Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T19:31:43.789931Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.06660"},"observation_digest":"sha256:5bf44d0aa7c5cc22cc0e79231742fdcd05ad6a438bc04290bb9c906afc13a173","observation_id":"78898fc8-66cf-4b12-b871-78e0c9f0cdf3","resolution":{"observed_at":"2026-08-11T19:31:43.789931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T19:29:52.601322Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.06673","last_updated":"2024-12-09T17:11:50Z","snapshot_observed_at":"2026-08-14T20:52:07.307688Z","submitted_at":"2024-12-09T17:11:50Z","title":"ILLUME: Illuminating Your LLMs to See, Draw, and Self-Enhance","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T19:29:52.601322Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.06673"},"observation_digest":"sha256:91f51018e5649ffa4bbc42141c828ac6ad6a560f0cc8451d02484b3d962a575e","observation_id":"06a20c3f-93ef-455b-9f24-5b6c4a46de20","resolution":{"observed_at":"2026-08-11T19:29:52.601322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T15:33:13.153399Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10958","last_updated":"2025-03-14T22:22:40Z","snapshot_observed_at":"2026-08-15T15:06:04.516299Z","submitted_at":"2024-12-14T20:29:29Z","title":"SoftVQ-VAE: Efficient 1-Dimensional Continuous Tokenizer","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:13.153399Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.10958"},"observation_digest":"sha256:a72798763abfd2aad3f4c8bd05c3f445c703ee10d29424b994f514d3124c777e","observation_id":"93142e31-20c6-43bc-96bf-f63d2547d81f","resolution":{"observed_at":"2026-08-11T15:33:13.153399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T14:39:28.390420Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.11767","last_updated":"2024-12-16T13:39:32Z","snapshot_observed_at":"2026-08-15T13:50:39.126462Z","submitted_at":"2024-12-16T13:39:32Z","title":"IDEA-Bench: How Far are Generative Models from Professional Designing?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T14:39:28.390420Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.11767"},"observation_digest":"sha256:a523dde0323248c45ed257efe8c3e051102bbac34ea54001bd4be3077f20917a","observation_id":"f155e2a1-a8a8-41b7-9cf0-7ce3355b4522","resolution":{"observed_at":"2026-08-11T14:39:28.390420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T14:01:05.296562Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12571","last_updated":"2024-12-17T06:03:05Z","snapshot_observed_at":"2026-08-14T06:16:00.801017Z","submitted_at":"2024-12-17T06:03:05Z","title":"ChatDiT: A Training-Free Baseline for Task-Agnostic Free-Form Chatting with Diffusion Transformers","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T14:01:05.296562Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.12571"},"observation_digest":"sha256:1f49cfa36227248e3f4efd6581aee075bb75628f75c76559e76da1b6ebfa0b63","observation_id":"c8b0ed51-e531-441f-914c-5fd9f51aba65","resolution":{"observed_at":"2026-08-11T14:01:05.296562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T11:37:41.952024Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15321","last_updated":"2025-03-19T06:16:54Z","snapshot_observed_at":"2026-08-14T22:45:56.271530Z","submitted_at":"2024-12-19T18:59:36Z","title":"Next Patch Prediction for Autoregressive Visual Generation","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T11:37:41.952024Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.15321"},"observation_digest":"sha256:015351c06d0db7d1b9e22c1a68edf14e60f045e59adb0c09688f053d27198366","observation_id":"50ed3788-2b69-4a11-939a-294476e8b714","resolution":{"observed_at":"2026-08-11T11:37:41.952024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T06:06:07.038110Z","title":"Making llama see and draw with seed tokenizer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16869","last_updated":"2024-12-22T05:42:40Z","snapshot_observed_at":"2026-08-14T10:29:20.071426Z","submitted_at":"2024-12-22T05:42:40Z","title":"CoF: Coarse to Fine-Grained Image Understanding for Multi-modal Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T06:06:07.038110Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.16869"},"observation_digest":"sha256:360359193ac758fcfe6a53ef30d15f0be9b045ce598860c7f61de3fb9668159a","observation_id":"c35e9009-05ac-4cb3-b827-e2ef6d93259d","resolution":{"observed_at":"2026-08-11T06:06:07.038110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-11T14:59:01.695742Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.18619","last_updated":"2024-12-30T03:00:30Z","snapshot_observed_at":"2026-08-13T14:56:43.704817Z","submitted_at":"2024-12-16T05:02:25Z","title":"Next Token Prediction Towards Multimodal Intelligence: A Comprehensive Survey","version":2},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-08-11T14:59:01.695742Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2412.18619"},"observation_digest":"sha256:fbae4f86279489bcab74df1c95ca945c28b2b6f60252706a18150275379679db","observation_id":"643a5392-fc0c-4d17-9e0a-683554bf27d4","resolution":{"observed_at":"2026-08-11T14:59:01.695742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-10T22:08:09.798294Z","title":"Making llama see and draw with seed tokenizer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-14T12:32:34.328935Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":232,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:09.798294Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:22b925fe837f265d642eb1eecc4532b4ab784d82233e0e67527274141cf35a7e","observation_id":"200f4aed-69d5-4c0e-8afd-77519d92a790","resolution":{"observed_at":"2026-08-10T22:08:09.798294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-08T05:54:15.899573Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08254","last_updated":"2025-02-12T09:49:43Z","snapshot_observed_at":"2026-08-15T10:30:50.174339Z","submitted_at":"2025-02-12T09:49:43Z","title":"UniCoRN: Unified Commented Retrieval Network with LMMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T05:54:15.899573Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2502.08254"},"observation_digest":"sha256:5bdf9c64ad0f2893752f92051599b749ab7e8a92bf9b7a622b75c2bc9090e613","observation_id":"4f9696d8-933d-4d91-8b84-0d05be0fc9a4","resolution":{"observed_at":"2026-08-08T05:54:15.899573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2503.14324","last_updated":"2026-04-20T17:55:12Z","snapshot_observed_at":"2026-08-02T12:34:37.994590Z","submitted_at":"2025-03-18T14:56:46Z","title":"DualToken: Towards Unifying Visual Understanding and Generation with Dual Visual Vocabularies","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T23:51:43.934329Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2503.14324"},"observation_digest":"sha256:fa12584c203e1d968e50310fe63f8884cee241870fe9304564498e23826b9157","observation_id":"063107a9-8985-4b3a-ad3b-6b88fed99714","resolution":{"observed_at":"2026-05-22T23:52:16.850621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-16T05:18:15.201350Z","title":"Making llama see and draw with seed tokenizer.arXiv preprint arXiv:2310.01218, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20996","last_updated":"2025-04-29T17:59:45Z","snapshot_observed_at":"2026-08-16T05:11:55.329193Z","submitted_at":"2025-04-29T17:59:45Z","title":"X-Fusion: Introducing New Modality to Frozen Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T05:18:15.201350Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2504.20996"},"observation_digest":"sha256:0d24ad0582c485d7bb8f8d6a4d2ec9a947aaf8dccfd2ac608ee60156de4cc457","observation_id":"e3e905b6-e91b-407b-91d8-2cccb9e2a468","resolution":{"observed_at":"2026-08-16T05:18:15.201350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-15T23:09:10.714439Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.05422","last_updated":"2025-08-15T08:56:27Z","snapshot_observed_at":"2026-08-16T05:36:45.421294Z","submitted_at":"2025-05-08T17:12:19Z","title":"TokLIP: Marry Visual Tokens to CLIP for Multimodal Comprehension and Generation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:09:10.714439Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.05422"},"observation_digest":"sha256:fdcb394d899a45ba2104e937b76ca1a816e216c33bd402f9508ceb94882182e9","observation_id":"e25a5b84-4a3e-43ed-ba33-4c57a1d4e3e7","resolution":{"observed_at":"2026-08-15T23:09:10.714439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2505.05472","last_updated":"2025-05-11T18:47:18Z","snapshot_observed_at":"2026-08-12T22:02:52.918587Z","submitted_at":"2025-05-08T17:58:57Z","title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T07:24:04.460276Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.05472"},"observation_digest":"sha256:d8f49aab8864c7f11eda45319fc1289a961f3836b06878eb57647c9b95cef68e","observation_id":"644111f9-966a-420c-b84c-94abc0b65c1e","resolution":{"observed_at":"2026-05-17T07:24:04.826174Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2505.17726","last_updated":"2026-05-19T09:55:01Z","snapshot_observed_at":"2026-08-14T12:18:58.071383Z","submitted_at":"2025-05-23T10:43:45Z","title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T02:06:35.204166Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.17726"},"observation_digest":"sha256:1df67778b88c874a33e0dd781b20cde19fd084fdc23c1fa453f987428e4f567e","observation_id":"a2f59b40-9cad-4d1e-be7b-b3962e0fecb5","resolution":{"observed_at":"2026-05-22T02:10:56.112562Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-07T14:04:54.366169Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20147","last_updated":"2025-07-24T05:31:36Z","snapshot_observed_at":"2026-08-08T06:11:49.265407Z","submitted_at":"2025-05-26T15:46:53Z","title":"FUDOKI: Discrete Flow-based Unified Understanding and Generation via Kinetic-Optimal Velocities","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:04:54.366169Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.20147"},"observation_digest":"sha256:5eaadcbcd158dd2b67195255fa7f6c7c407dd75b070e7be13d978027d716f46d","observation_id":"11016a83-596a-4a6e-bee4-aad63d187c44","resolution":{"observed_at":"2026-08-07T14:04:54.366169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2505.21374","last_updated":"2025-05-27T16:05:01Z","snapshot_observed_at":"2026-08-15T17:05:32.253271Z","submitted_at":"2025-05-27T16:05:01Z","title":"Video-Holmes: Can MLLM Think Like Holmes for Complex Video Reasoning?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T05:40:55.944288Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.21374"},"observation_digest":"sha256:39db5a7c665afd4bc18c4c4f87c8a363edcde33e85226c45d5f677ae9af68b53","observation_id":"cda3c8cc-5278-4f38-8deb-67b5537f8dae","resolution":{"observed_at":"2026-05-17T05:40:55.995559Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-07T12:45:06.140712Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23656","last_updated":"2025-05-29T17:06:44Z","snapshot_observed_at":"2026-08-15T14:35:51.945342Z","submitted_at":"2025-05-29T17:06:44Z","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:06.140712Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2505.23656"},"observation_digest":"sha256:800749cd995e654d5bef91506a251b04ad750a0af431c9607298dcfb006ca318","observation_id":"24ac5404-754b-4718-b8f3-717d54631f7d","resolution":{"observed_at":"2026-08-07T12:45:06.140712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-07T10:23:12.502691Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05501","last_updated":"2025-06-05T18:36:33Z","snapshot_observed_at":"2026-08-13T05:25:22.532423Z","submitted_at":"2025-06-05T18:36:33Z","title":"FocusDiff: Advancing Fine-Grained Text-Image Alignment for Autoregressive Visual Generation through RL","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T10:23:12.502691Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2506.05501"},"observation_digest":"sha256:841f218df096973bdb294219c3df86d1867a4a569e79b0d81513cc7d8d2a627a","observation_id":"d3487a38-dc62-4b86-bca4-cf93de2dbdb1","resolution":{"observed_at":"2026-08-07T10:23:12.502691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-15T18:46:10.655820Z","title":"Making llama see and draw with seed tokenizer.arXiv preprint arXiv:2310.01218, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18898","last_updated":"2025-06-23T17:59:14Z","snapshot_observed_at":"2026-08-15T23:43:47.561047Z","submitted_at":"2025-06-23T17:59:14Z","title":"Vision as a Dialect: Unifying Visual Understanding and Generation via Text-Aligned Representations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T18:46:10.655820Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2506.18898"},"observation_digest":"sha256:077eed9aa08cbfdbf283eead6193f1030dbbe1c3eca41415c25c99de040ea52b","observation_id":"87f0b4e0-468c-4b1e-8e6f-67262c9afe7f","resolution":{"observed_at":"2026-08-15T18:46:10.655820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-06T17:47:41.979247Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09910","last_updated":"2025-07-14T04:31:15Z","snapshot_observed_at":"2026-08-09T06:49:52.603374Z","submitted_at":"2025-07-14T04:31:15Z","title":"IGD: Instructional Graphic Design with Multimodal Layer Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:47:41.979247Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2507.09910"},"observation_digest":"sha256:34f0fea74232f1c338b9605a905635424b6ac27eafe1f818ae758c3a588da2cf","observation_id":"05fce673-ba5d-4366-9c62-e29eeb872097","resolution":{"observed_at":"2026-08-06T17:47:41.979247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-05T22:35:50.726451Z","title":"Making llama see and draw with seed tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.06895","last_updated":"2025-08-09T09:00:45Z","snapshot_observed_at":"2026-08-12T16:47:37.700496Z","submitted_at":"2025-08-09T09:00:45Z","title":"BASIC: Boosting Visual Alignment with Intrinsic Refined Embeddings in Multimodal Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T22:35:50.726451Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2508.06895"},"observation_digest":"sha256:5e0faf2a96db37ea090aa325e0192118b68b68340234a68e42570165218cf63e","observation_id":"0035c43f-fdc4-476a-b219-aceb791b2b21","resolution":{"observed_at":"2026-08-05T22:35:50.726451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-05T21:41:11.885994Z","title":"Making llama see and draw with seed tokenizer.arXiv preprint arXiv:2310.01218,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.08098","last_updated":"2025-08-14T17:38:47Z","snapshot_observed_at":"2026-08-16T05:34:04.126153Z","submitted_at":"2025-08-11T15:37:22Z","title":"TBAC-UniImage: Unified Understanding and Generation by Ladder-Side Diffusion Tuning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T21:41:11.885994Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2508.08098"},"observation_digest":"sha256:36d733c07ecc05139cfc34b80433aa02b9463ae94ad9ff250bdadd15b65c7dd5","observation_id":"a97b6fa1-1915-4ead-9d96-76dff61ce866","resolution":{"observed_at":"2026-08-05T21:41:11.885994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-05T06:00:27.412176Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.04606","last_updated":"2025-09-04T18:41:59Z","snapshot_observed_at":"2026-08-15T23:11:42.439873Z","submitted_at":"2025-09-04T18:41:59Z","title":"Sample-efficient Integration of New Modalities into Large Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T06:00:27.412176Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2509.04606"},"observation_digest":"sha256:cb52ce9340ee02163dc9641abe6e3bf2960b9e70e2b01bac74a6bcb8ec69bc59","observation_id":"27c96fe5-b1de-46d0-b399-c6852e091228","resolution":{"observed_at":"2026-08-05T06:00:27.412176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-04T15:45:10.086435Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.18588","last_updated":"2026-06-17T07:22:49Z","snapshot_observed_at":"2026-08-10T12:23:09.024771Z","submitted_at":"2025-09-23T03:15:53Z","title":"UniECG: Understanding and Generating ECG in One Unified Model","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T15:45:10.086435Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2509.18588"},"observation_digest":"sha256:2a37d8e562c6a9f17cbe794c28b6bb025c97ec80527b687bd39295e9a25ec879","observation_id":"94b9f1ac-59f5-48b0-99f9-c221d39aee55","resolution":{"observed_at":"2026-08-04T15:45:10.086435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-03T03:57:31.361767Z","title":"Making llama see and draw with seed tokenizer.arXiv preprint arXiv:2310.01218,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.06442","last_updated":"2026-06-01T15:54:20Z","snapshot_observed_at":"2026-08-12T20:50:12.933006Z","submitted_at":"2026-02-06T07:11:50Z","title":"ChatUMM: Robust Context Tracking for Conversational Interleaved Generation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T03:57:31.361767Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2602.06442"},"observation_digest":"sha256:a749115abdade92a9b1e5daf5d4df14db057aa30cbf21c2da16d517ff346c2ba","observation_id":"05cd402b-5186-48e7-9b3e-bded583af4b3","resolution":{"observed_at":"2026-08-03T03:57:31.361767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2605.00503","last_updated":"2026-05-04T10:19:31Z","snapshot_observed_at":"2026-08-16T05:55:48.478648Z","submitted_at":"2026-05-01T08:25:51Z","title":"End-to-End Autoregressive Image Generation with 1D Semantic Tokenizer","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T19:41:03.302303Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2605.00503"},"observation_digest":"sha256:3dccda4cc343024e65b89d5fb3d5f35e57e9bdaa4b59968d37ec7ed38091580b","observation_id":"d778adb8-b3db-4394-b248-abdbd10180b3","resolution":{"observed_at":"2026-05-11T15:36:05.964626Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:e256a34782700fedfb3bb57a64a5efc7257bf44b4e84f6eb67025f255a609c08","observation_id":"f43ffe58-d574-4613-868b-02f1b8e1feb9","resolution":{"observed_at":"2026-05-13T05:17:18.821835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.07171","last_updated":"2026-06-05T11:40:03Z","snapshot_observed_at":"2026-08-02T22:25:34.082799Z","submitted_at":"2026-06-05T11:40:03Z","title":"When Recovery Matters: The Blind Spot of Surrogate Privacy in MLLM Editing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T22:08:57.792229Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.07171"},"observation_digest":"sha256:6e3043b758a8160b45fbdeb550d6706f4cadeace3cca0433ecbfb2b04962e1c2","observation_id":"186a5098-c54c-4649-8fac-6b9f4cbc284c","resolution":{"observed_at":"2026-07-02T17:07:13.122612Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.13289","last_updated":"2026-06-11T12:46:07Z","snapshot_observed_at":"2026-08-12T18:02:55.540174Z","submitted_at":"2026-06-11T12:46:07Z","title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-06-27T07:01:07.362430Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.13289"},"observation_digest":"sha256:d1103f8a19ff5571a024ebfb3069baf5324b5f8bb45ba244012b790f70e70a2a","observation_id":"7805d954-f801-4294-a566-f745f9568dd5","resolution":{"observed_at":"2026-07-03T14:38:28.919623Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.13679","last_updated":"2026-06-11T17:59:50Z","snapshot_observed_at":"2026-08-12T10:38:27.260916Z","submitted_at":"2026-06-11T17:59:50Z","title":"InterleaveThinker: Reinforcing Agentic Interleaved Generation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T06:42:34.126336Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.13679"},"observation_digest":"sha256:7d340ffcbfd842494caf49db4900074f71ea00bf8c2e9525cc9fd8a6489134b2","observation_id":"8a32fd32-f805-420a-a075-5372a1a9d8d8","resolution":{"observed_at":"2026-07-03T15:08:32.891323Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-12T17:43:10.201466Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":2},"reference_index":253,"source":"arxiv_source","source_observed_at":"2026-06-26T09:19:50.623741Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:b5c4004f971cab8edf05e631c2d861312ed6915b49692a21979eb52fa3694ab3","observation_id":"f42d80b3-3afe-45bd-8138-7db669b41d8f","resolution":{"observed_at":"2026-07-04T09:59:44.766332Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-12T17:43:10.201466Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":3},"reference_index":252,"source":"arxiv_source","source_observed_at":"2026-06-29T01:18:19.195007Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:da22f6bfc0c4d4351db0ad327a9ddbd75e27b666fd4c91871d289a4739822f54","observation_id":"1208ccc4-baff-48ac-b94c-2a2fb21b9c1a","resolution":{"observed_at":"2026-07-01T18:55:59.718833Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2606.30054","last_updated":"2026-06-29T09:45:15Z","snapshot_observed_at":"2026-08-14T12:34:00.326381Z","submitted_at":"2026-06-29T09:45:15Z","title":"Illuminating Unified Multimodal Model for Free-form Interleaved Text-Image Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-30T06:04:07.327934Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2606.30054"},"observation_digest":"sha256:cedf2ca18dddd6d60d5d575da9d843b5957e828f0c1173c9c53af27c368e0b02","observation_id":"a4de679e-c1e1-4667-ac3b-b459f967aebe","resolution":{"observed_at":"2026-06-30T06:04:20.951281Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-12T06:13:16.894125Z","title":"arXiv preprint arXiv:2310.01218 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02907","last_updated":"2026-07-03T03:14:11Z","snapshot_observed_at":"2026-08-13T04:12:16.138751Z","submitted_at":"2026-07-03T03:14:11Z","title":"ProLaViT: Learning Progressive Latent Visual Thoughts in Structured Latent Space","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-12T06:13:16.894125Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2607.02907"},"observation_digest":"sha256:97af347be99b354ec32a98d8721d0c20435a47eda6791da3be9a89663922daf1","observation_id":"7c8cef05-b7ad-465a-a62c-34c53062e132","resolution":{"observed_at":"2026-07-12T06:13:16.894125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-12T01:50:59.184754Z","title":"arXiv preprint arXiv:2310.01218 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03530","last_updated":"2026-07-03T17:59:58Z","snapshot_observed_at":"2026-08-15T02:40:10.141978Z","submitted_at":"2026-07-03T17:59:58Z","title":"MentalThink: Shaping Thoughts in Mental SVG World","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-07-12T01:50:59.184754Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2607.03530"},"observation_digest":"sha256:0d52924e314681877f8d7f74ff2748c93684780d87993ce98bb27016b6879d09","observation_id":"a434d4f3-337d-46d7-978f-8e0dea4dad31","resolution":{"observed_at":"2026-07-12T01:50:59.184754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2607.07117","last_updated":"2026-07-08T07:58:48Z","snapshot_observed_at":"2026-08-14T18:43:29.717220Z","submitted_at":"2026-07-08T07:58:48Z","title":"Tree-of-Thoughts Reasoning for Text-to-Image In-Context Learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-09T19:57:27.683657Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2607.07117"},"observation_digest":"sha256:8eb7ddfd4f69c9cc427b60c0943e07bbedfa2d0e042902dff712a99c5bcae92c","observation_id":"59d22b61-4684-47c4-b9ba-0d776255867f","resolution":{"observed_at":"2026-07-09T20:16:29.589913Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-08-01T04:29:48.538845Z","title":"arXiv preprint arXiv:2310.01218 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22531","last_updated":"2026-07-24T17:59:39Z","snapshot_observed_at":"2026-08-14T14:09:07.258818Z","submitted_at":"2026-07-24T17:59:39Z","title":"Twins: Learn to Predict Unified Representations with Focal Loss","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-01T04:29:48.538845Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2607.22531"},"observation_digest":"sha256:b52ae5bf923b3ac7265ece0a362654514beca4bedcc5dff62c394f0f0a34ac99","observation_id":"7ff494b4-408f-4432-80d2-6df5a068c518","resolution":{"observed_at":"2026-08-01T04:29:48.538845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.01218/citation-record","integrity":"/paper/2310.01218/integrity","json":"/paper/2310.01218/citation-record.json","paper":"/paper/2310.01218"},"outbound":[],"paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T07:40:15.155512Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 50 inbound Pith citation observations for arXiv:2310.01218."}