{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B2LNTAXPLOP4SY2ATU66JTVS7Z","short_pith_number":"pith:B2LNTAXP","schema_version":"1.0","canonical_sha256":"0e96d982ef5b9fc963409d3de4ceb2fe406e070af7a0e6e9ae7d8b09a23a4033","source":{"kind":"arxiv","id":"2404.10864","version":1},"attestation_state":"computed","paper":{"title":"Vocabulary-free Image Classification and Semantic Segmentation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alessandro Conti, Elisa Ricci, Enrico Fini, Massimiliano Mancini, Paolo Rota, Yiming Wang","submitted_at":"2024-04-16T19:27:21Z","abstract_excerpt":"Large vision-language models revolutionized image classification and semantic segmentation paradigms. However, they typically assume a pre-defined set of categories, or vocabulary, at test time for composing textual prompts. This assumption is impractical in scenarios with unknown or evolving semantic context. Here, we address this issue and introduce the Vocabulary-free Image Classification (VIC) task, which aims to assign a class from an unconstrained language-induced semantic space to an input image without needing a known vocabulary. VIC is challenging due to the vastness of the semantic s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.10864","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-16T19:27:21Z","cross_cats_sorted":[],"title_canon_sha256":"b7218449a1982ee4c984a250ecd3d7a81f0b40ffd574cb16699bc6ab0608156b","abstract_canon_sha256":"bdf1a8369b2d102eafa05b7520f37b38ac548ad1b9c77e2411a339ea949304bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:09:00.054448Z","signature_b64":"bkIk8AclvkpzCcgL0vUsdMMtuL7RIgOw74zrMug/DXLRYbqNlvn9gPmJ1QtLnVmpc67bIdHFxAEjlvrIkfblAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e96d982ef5b9fc963409d3de4ceb2fe406e070af7a0e6e9ae7d8b09a23a4033","last_reissued_at":"2026-07-05T08:09:00.054001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:09:00.054001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vocabulary-free Image Classification and Semantic Segmentation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alessandro Conti, Elisa Ricci, Enrico Fini, Massimiliano Mancini, Paolo Rota, Yiming Wang","submitted_at":"2024-04-16T19:27:21Z","abstract_excerpt":"Large vision-language models revolutionized image classification and semantic segmentation paradigms. However, they typically assume a pre-defined set of categories, or vocabulary, at test time for composing textual prompts. This assumption is impractical in scenarios with unknown or evolving semantic context. Here, we address this issue and introduce the Vocabulary-free Image Classification (VIC) task, which aims to assign a class from an unconstrained language-induced semantic space to an input image without needing a known vocabulary. VIC is challenging due to the vastness of the semantic s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.10864","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.10864/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.10864","created_at":"2026-07-05T08:09:00.054057+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.10864v1","created_at":"2026-07-05T08:09:00.054057+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.10864","created_at":"2026-07-05T08:09:00.054057+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2LNTAXPLOP4","created_at":"2026-07-05T08:09:00.054057+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2LNTAXPLOP4SY2A","created_at":"2026-07-05T08:09:00.054057+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2LNTAXP","created_at":"2026-07-05T08:09:00.054057+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05547","citing_title":"Adapting Vision-Language Models Without Labels: A Comprehensive Survey","ref_index":182,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z","json":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z.json","graph_json":"https://pith.science/api/pith-number/B2LNTAXPLOP4SY2ATU66JTVS7Z/graph.json","events_json":"https://pith.science/api/pith-number/B2LNTAXPLOP4SY2ATU66JTVS7Z/events.json","paper":"https://pith.science/paper/B2LNTAXP"},"agent_actions":{"view_html":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z","download_json":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z.json","view_paper":"https://pith.science/paper/B2LNTAXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.10864&json=true","fetch_graph":"https://pith.science/api/pith-number/B2LNTAXPLOP4SY2ATU66JTVS7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/B2LNTAXPLOP4SY2ATU66JTVS7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z/action/storage_attestation","attest_author":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z/action/author_attestation","sign_citation":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z/action/citation_signature","submit_replication":"https://pith.science/pith/B2LNTAXPLOP4SY2ATU66JTVS7Z/action/replication_record"}},"created_at":"2026-07-05T08:09:00.054057+00:00","updated_at":"2026-07-05T08:09:00.054057+00:00"}