{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z5BRNKDL4C7XIPIK5ZCBSW4DD5","short_pith_number":"pith:Z5BRNKDL","schema_version":"1.0","canonical_sha256":"cf4316a86be0bf743d0aee44195b831f4f30c7743e8f655a46f3d8531e1ebf87","source":{"kind":"arxiv","id":"2411.14347","version":3},"attestation_state":"computed","paper":{"title":"DINO-X: A Unified Vision Model for Open-World Object Detection and Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Han Gao, Hao Zhang, Hongjie Huang, Junyi Shen, Kent Yu, Lei Zhang, Qing Jiang, Shilong Liu, Tianhe Ren, Wenlong Liu, Xiaoke Jiang, Xingyu Chen, Yihao Chen, Yuan Gao, Yuda Xiong, Yuhong Zhang, Zhaoyang Zeng, Zhengyu Ma, Zhuheng Song","submitted_at":"2024-11-21T17:42:20Z","abstract_excerpt":"In this paper, we introduce DINO-X, which is a unified object-centric vision model developed by IDEA Research with the best open-world object detection performance to date. DINO-X employs the same Transformer-based encoder-decoder architecture as Grounding DINO 1.5 to pursue an object-level representation for open-world object understanding. To make long-tailed object detection easy, DINO-X extends its input options to support text prompt, visual prompt, and customized prompt. With such flexible prompt options, we develop a universal object prompt to support prompt-free open-world detection, m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14347","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-21T17:42:20Z","cross_cats_sorted":[],"title_canon_sha256":"8c60b7fcb031d2896812c1acb5018d93961233667dc86b6884184b029911e3d1","abstract_canon_sha256":"2709e8aacf57ad6b0c942b81301f498e980b677676494f074c1d7adf20779336"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:18.569517Z","signature_b64":"UOcuEyiW5emOj8FMRdEnSxqoiXEnkXsQAkExSYP+JH7vpFZ4TnnOfsa9rHGtb3EQPo/+CRhnqc1hvTdZt7TTCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf4316a86be0bf743d0aee44195b831f4f30c7743e8f655a46f3d8531e1ebf87","last_reissued_at":"2026-07-05T11:03:18.569004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:18.569004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DINO-X: A Unified Vision Model for Open-World Object Detection and Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Han Gao, Hao Zhang, Hongjie Huang, Junyi Shen, Kent Yu, Lei Zhang, Qing Jiang, Shilong Liu, Tianhe Ren, Wenlong Liu, Xiaoke Jiang, Xingyu Chen, Yihao Chen, Yuan Gao, Yuda Xiong, Yuhong Zhang, Zhaoyang Zeng, Zhengyu Ma, Zhuheng Song","submitted_at":"2024-11-21T17:42:20Z","abstract_excerpt":"In this paper, we introduce DINO-X, which is a unified object-centric vision model developed by IDEA Research with the best open-world object detection performance to date. DINO-X employs the same Transformer-based encoder-decoder architecture as Grounding DINO 1.5 to pursue an object-level representation for open-world object understanding. To make long-tailed object detection easy, DINO-X extends its input options to support text prompt, visual prompt, and customized prompt. With such flexible prompt options, we develop a universal object prompt to support prompt-free open-world detection, m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14347","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14347/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14347","created_at":"2026-07-05T11:03:18.569064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14347v3","created_at":"2026-07-05T11:03:18.569064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14347","created_at":"2026-07-05T11:03:18.569064+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z5BRNKDL4C7X","created_at":"2026-07-05T11:03:18.569064+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z5BRNKDL4C7XIPIK","created_at":"2026-07-05T11:03:18.569064+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z5BRNKDL","created_at":"2026-07-05T11:03:18.569064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08541","citing_title":"VocaDet: Sample-Driven Open-Vocabulary Object Detection and Segmentation via Visual Tokenization and Vector Database Retrieval","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11546","citing_title":"VL-DINO: Leveraging CLIP Vision-Language Knowledge for Open-Vocabulary Object Detectio","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00338","citing_title":"DroneFINE: Domain-Aware Parameter-Efficient Fine-Tuning of Vision-Language Detectors for Drone Images","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00816","citing_title":"Towards High-Resolution Visual Perception via Hierarchical Entity Exploration","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05635","citing_title":"ShotCrop$^3$: Cropping Human-Centric Images into Cinematic Triple-Shot Compositions","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09882","citing_title":"WHU-Infra3D: A Full-stack Multi-modal Dataset and Benchmark for 3D Roadside Infrastructure Inventory","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14795","citing_title":"COAL: Counterfactual and Observation-Enhanced Alignment Learning for Discriminative Referring Multi-Object Tracking","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14923","citing_title":"SceneParser: Hierarchical Scene Parsing for Visual Semantics Understanding","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20329","citing_title":"Image Generators are Generalist Vision Learners","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20329","citing_title":"Image Generators are Generalist Vision Learners","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13292","citing_title":"See&Say: Vision Language Guided Safe Zone Detection for Autonomous Package Delivery Drones","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5","json":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5.json","graph_json":"https://pith.science/api/pith-number/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/graph.json","events_json":"https://pith.science/api/pith-number/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/events.json","paper":"https://pith.science/paper/Z5BRNKDL"},"agent_actions":{"view_html":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5","download_json":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5.json","view_paper":"https://pith.science/paper/Z5BRNKDL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14347&json=true","fetch_graph":"https://pith.science/api/pith-number/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/graph.json","fetch_events":"https://pith.science/api/pith-number/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/action/storage_attestation","attest_author":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/action/author_attestation","sign_citation":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/action/citation_signature","submit_replication":"https://pith.science/pith/Z5BRNKDL4C7XIPIK5ZCBSW4DD5/action/replication_record"}},"created_at":"2026-07-05T11:03:18.569064+00:00","updated_at":"2026-07-05T11:03:18.569064+00:00"}