{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LBJMNDGG6A2OUJGZRP6JESBNBB","short_pith_number":"pith:LBJMNDGG","schema_version":"1.0","canonical_sha256":"5852c68cc6f034ea24d98bfc92482d0861bfa5f99ab6c7a06bc216967caf5e72","source":{"kind":"arxiv","id":"2307.08041","version":2},"attestation_state":"computed","paper":{"title":"Planting a SEED of Vision in Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Xintao Wang, Ying Shan, Yixiao Ge, Yuying Ge, Ziyun Zeng","submitted_at":"2023-07-16T13:41:39Z","abstract_excerpt":"We present SEED, an elaborate image tokenizer that empowers Large Language Models (LLMs) with the emergent ability to SEE and Draw at the same time. Research on image tokenizers has previously reached an impasse, as frameworks employing quantized visual tokens have lost prominence due to subpar performance and convergence in multimodal comprehension (compared to BLIP-2, etc.) or generation (compared to Stable Diffusion, etc.). Despite the limitations, we remain confident in its natural capacity to unify visual and textual representations, facilitating scalable multimodal training with LLM's or"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.08041","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-16T13:41:39Z","cross_cats_sorted":[],"title_canon_sha256":"cdcf49fee72618cfb9454ef9ae832a4faa4dc302b4ddabd9b4d195ad37284738","abstract_canon_sha256":"96596641835a600651b90345a15feac75efdf99b94c52dc13eb2bb86ac7d069a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:40:35.340390Z","signature_b64":"pAK3yWR26a92AKbfdGvNwzFGkq9Gc60GGuVGrLkpjmZ+v/86dyKc4Hkdd6ZsTJGhWhAKFccjHDO25GNkqeFlBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5852c68cc6f034ea24d98bfc92482d0861bfa5f99ab6c7a06bc216967caf5e72","last_reissued_at":"2026-07-05T06:40:35.339866Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:40:35.339866Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Planting a SEED of Vision in Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Xintao Wang, Ying Shan, Yixiao Ge, Yuying Ge, Ziyun Zeng","submitted_at":"2023-07-16T13:41:39Z","abstract_excerpt":"We present SEED, an elaborate image tokenizer that empowers Large Language Models (LLMs) with the emergent ability to SEE and Draw at the same time. Research on image tokenizers has previously reached an impasse, as frameworks employing quantized visual tokens have lost prominence due to subpar performance and convergence in multimodal comprehension (compared to BLIP-2, etc.) or generation (compared to Stable Diffusion, etc.). Despite the limitations, we remain confident in its natural capacity to unify visual and textual representations, facilitating scalable multimodal training with LLM's or"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.08041","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.08041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.08041","created_at":"2026-07-05T06:40:35.339927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.08041v2","created_at":"2026-07-05T06:40:35.339927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.08041","created_at":"2026-07-05T06:40:35.339927+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBJMNDGG6A2O","created_at":"2026-07-05T06:40:35.339927+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBJMNDGG6A2OUJGZ","created_at":"2026-07-05T06:40:35.339927+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBJMNDGG","created_at":"2026-07-05T06:40:35.339927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07171","citing_title":"When Recovery Matters: The Blind Spot of Surrogate Privacy in MLLM Editing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2308.08089","citing_title":"DragNUWA: Fine-grained Control in Video Generation by Integrating Text, Image, and Trajectory","ref_index":258,"is_internal_anchor":false},{"citing_arxiv_id":"2506.04565","citing_title":"From Standalone LLMs to Integrated Intelligence: A Survey of Compound Al Systems","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":216,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05472","citing_title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04429","citing_title":"VILA-U: a Unified Foundation Model Integrating Visual Understanding and Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14396","citing_title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13848","citing_title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2504.06256","citing_title":"Transfer between Modalities with MetaQueries","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13080","citing_title":"Learning to See What You Need: Gaze Attention for Multimodal Large Language Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2307.16125","citing_title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10564","citing_title":"DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving","ref_index":206,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB","json":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB.json","graph_json":"https://pith.science/api/pith-number/LBJMNDGG6A2OUJGZRP6JESBNBB/graph.json","events_json":"https://pith.science/api/pith-number/LBJMNDGG6A2OUJGZRP6JESBNBB/events.json","paper":"https://pith.science/paper/LBJMNDGG"},"agent_actions":{"view_html":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB","download_json":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB.json","view_paper":"https://pith.science/paper/LBJMNDGG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.08041&json=true","fetch_graph":"https://pith.science/api/pith-number/LBJMNDGG6A2OUJGZRP6JESBNBB/graph.json","fetch_events":"https://pith.science/api/pith-number/LBJMNDGG6A2OUJGZRP6JESBNBB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB/action/storage_attestation","attest_author":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB/action/author_attestation","sign_citation":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB/action/citation_signature","submit_replication":"https://pith.science/pith/LBJMNDGG6A2OUJGZRP6JESBNBB/action/replication_record"}},"created_at":"2026-07-05T06:40:35.339927+00:00","updated_at":"2026-07-05T06:40:35.339927+00:00"}