{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3UDWR44377HKAEPHL5XUHZCVYL","short_pith_number":"pith:3UDWR443","schema_version":"1.0","canonical_sha256":"dd0768f39bffcea011e75f6f43e455c2cb64389dc06817ac5d8b552da14f8f4f","source":{"kind":"arxiv","id":"2302.08113","version":1},"attestation_state":"computed","paper":{"title":"MultiDiffusion: Fusing Diffusion Paths for Controlled Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Lior Yariv, Omer Bar-Tal, Tali Dekel, Yaron Lipman","submitted_at":"2023-02-16T06:28:29Z","abstract_excerpt":"Recent advances in text-to-image generation with diffusion models present transformative capabilities in image quality. However, user controllability of the generated image, and fast adaptation to new tasks still remains an open challenge, currently mostly addressed by costly and long re-training and fine-tuning or ad-hoc adaptations to specific image generation tasks. In this work, we present MultiDiffusion, a unified framework that enables versatile and controllable image generation, using a pre-trained text-to-image diffusion model, without any further training or finetuning. At the center "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08113","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-02-16T06:28:29Z","cross_cats_sorted":[],"title_canon_sha256":"f44ca57b6a70d00785b41109ea923df9f1682372f87b889260c168f957da3625","abstract_canon_sha256":"1f68db2ac98589f0f1a33f05a9c1f761c2a0b65acf96879118dc1a02bab4c524"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:34.508932Z","signature_b64":"GGpCuRrLkvLi+cCGdI1e9t4tslC3o/pPslCXiNBtF6iJ1rP0eQcCq+mvYeHu2ajCxz4P99OO2186gLKP+B2kAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd0768f39bffcea011e75f6f43e455c2cb64389dc06817ac5d8b552da14f8f4f","last_reissued_at":"2026-07-05T05:42:34.508436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:34.508436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MultiDiffusion: Fusing Diffusion Paths for Controlled Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Lior Yariv, Omer Bar-Tal, Tali Dekel, Yaron Lipman","submitted_at":"2023-02-16T06:28:29Z","abstract_excerpt":"Recent advances in text-to-image generation with diffusion models present transformative capabilities in image quality. However, user controllability of the generated image, and fast adaptation to new tasks still remains an open challenge, currently mostly addressed by costly and long re-training and fine-tuning or ad-hoc adaptations to specific image generation tasks. In this work, we present MultiDiffusion, a unified framework that enables versatile and controllable image generation, using a pre-trained text-to-image diffusion model, without any further training or finetuning. At the center "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08113","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08113/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08113","created_at":"2026-07-05T05:42:34.508499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08113v1","created_at":"2026-07-05T05:42:34.508499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08113","created_at":"2026-07-05T05:42:34.508499+00:00"},{"alias_kind":"pith_short_12","alias_value":"3UDWR44377HK","created_at":"2026-07-05T05:42:34.508499+00:00"},{"alias_kind":"pith_short_16","alias_value":"3UDWR44377HKAEPH","created_at":"2026-07-05T05:42:34.508499+00:00"},{"alias_kind":"pith_short_8","alias_value":"3UDWR443","created_at":"2026-07-05T05:42:34.508499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05392","citing_title":"SynCity 3000: Bootstrapping Scene-Scale 3D Diffusion","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30512","citing_title":"PhyDrawGen: Physically Grounded Diagram Generation from Natural Language","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00837","citing_title":"Coarse-to-Fine Compositional Diffusion for Long-Horizon Planning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23888","citing_title":"GenRecon: Bridging Generative Priors for Multi-View 3D Scene Reconstruction","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21766","citing_title":"BodyReLux: Temporally Consistent Full-Body Video Relighting","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2603.21002","citing_title":"SURF: Signature-Retained Fast Video Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17543","citing_title":"HL-OutPaint: Coarse-to-Fine Video Outpainting for High-Resolution Long-Range Videos","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2307.10373","citing_title":"TokenFlow: Consistent Diffusion Features for Consistent Video Editing","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08309","citing_title":"InfiniteDiffusion: Bridging Learned Fidelity and Procedural Utility for Open-World Terrain Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2302.05543","citing_title":"Adding Conditional Control to Text-to-Image Diffusion Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05127","citing_title":"LooseRoPE: Content-aware Attention Manipulation for Semantic Harmonization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2309.03453","citing_title":"SyncDreamer: Generating Multiview-consistent Images from a Single-view Image","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03342","citing_title":"Tiled Prompts: Overcoming Prompt Misguidance in Image and Video Super-Resolution","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08698","citing_title":"Supersampling Stable Diffusion and Beyond: A Seamless, Training-Free Approach for Scaling Neural Networks Using Common Interpolation Methods","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14337","citing_title":"IG-Diff: Complex Night Scene Restoration with Illumination-Guided Diffusion Model","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03359","citing_title":"Mix3R: Mixing Feed-forward Reconstruction and Generative 3D Priors for Joint Multi-view Aligned 3D Reconstruction and Pose Estimation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09622","citing_title":"Any2Any 3D Diffusion Models with Knowledge Transfer: A Radiotherapy Planning Study","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08698","citing_title":"Supersampling Stable Diffusion and Beyond: A Seamless, Training-Free Approach for Scaling Neural Networks Using Common Interpolation Methods","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2403.05135","citing_title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18258","citing_title":"Long-Text-to-Image Generation via Compositional Prompt Decomposition","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL","json":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL.json","graph_json":"https://pith.science/api/pith-number/3UDWR44377HKAEPHL5XUHZCVYL/graph.json","events_json":"https://pith.science/api/pith-number/3UDWR44377HKAEPHL5XUHZCVYL/events.json","paper":"https://pith.science/paper/3UDWR443"},"agent_actions":{"view_html":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL","download_json":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL.json","view_paper":"https://pith.science/paper/3UDWR443","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08113&json=true","fetch_graph":"https://pith.science/api/pith-number/3UDWR44377HKAEPHL5XUHZCVYL/graph.json","fetch_events":"https://pith.science/api/pith-number/3UDWR44377HKAEPHL5XUHZCVYL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL/action/storage_attestation","attest_author":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL/action/author_attestation","sign_citation":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL/action/citation_signature","submit_replication":"https://pith.science/pith/3UDWR44377HKAEPHL5XUHZCVYL/action/replication_record"}},"created_at":"2026-07-05T05:42:34.508499+00:00","updated_at":"2026-07-05T05:42:34.508499+00:00"}