{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F6SVHGEPFUGZ2USARYN4JM4YCV","short_pith_number":"pith:F6SVHGEP","schema_version":"1.0","canonical_sha256":"2fa553988f2d0d9d52408e1bc4b3981554021fbfb737cba1137643ae7dc98178","source":{"kind":"arxiv","id":"2404.12652","version":2},"attestation_state":"computed","paper":{"title":"Pre-trained Vision-Language Models Learn Discoverable Visual Concepts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chen Sun, Hao Tan, Tian Yun, Trung Bui, Yuan Zang","submitted_at":"2024-04-19T06:41:32Z","abstract_excerpt":"Do vision-language models (VLMs) pre-trained to caption an image of a \"durian\" learn visual concepts such as \"brown\" (color) and \"spiky\" (texture) at the same time? We aim to answer this question as visual concepts learned \"for free\" would enable wide applications such as neuro-symbolic reasoning or human-interpretable object classification. We assume that the visual concepts, if captured by pre-trained VLMs, can be extracted by their vision-language interface with text-based concept prompts. We observe that recent works prompting VLMs with concepts often differ in their strategies to define a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12652","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-19T06:41:32Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"03532a2771bd67e0e294c978a7a8931501997ef48494e19b54cc9851121e4b8e","abstract_canon_sha256":"20c2fb17bad2bf17d5300f11db569a4648edca51b0eb96968a63f57652838e47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:36.161600Z","signature_b64":"Wnqm6vKt2lrO5oFECjyh7Lo8aFfPZnIc2k1YGzRA1XFMYHNjZJyxaAqmo8ZV7SRwD6l+I6/M26hwFV8BTRldCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fa553988f2d0d9d52408e1bc4b3981554021fbfb737cba1137643ae7dc98178","last_reissued_at":"2026-07-05T10:00:36.161161Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:36.161161Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-trained Vision-Language Models Learn Discoverable Visual Concepts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chen Sun, Hao Tan, Tian Yun, Trung Bui, Yuan Zang","submitted_at":"2024-04-19T06:41:32Z","abstract_excerpt":"Do vision-language models (VLMs) pre-trained to caption an image of a \"durian\" learn visual concepts such as \"brown\" (color) and \"spiky\" (texture) at the same time? We aim to answer this question as visual concepts learned \"for free\" would enable wide applications such as neuro-symbolic reasoning or human-interpretable object classification. We assume that the visual concepts, if captured by pre-trained VLMs, can be extracted by their vision-language interface with text-based concept prompts. We observe that recent works prompting VLMs with concepts often differ in their strategies to define a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12652","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12652/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12652","created_at":"2026-07-05T10:00:36.161217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12652v2","created_at":"2026-07-05T10:00:36.161217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12652","created_at":"2026-07-05T10:00:36.161217+00:00"},{"alias_kind":"pith_short_12","alias_value":"F6SVHGEPFUGZ","created_at":"2026-07-05T10:00:36.161217+00:00"},{"alias_kind":"pith_short_16","alias_value":"F6SVHGEPFUGZ2USA","created_at":"2026-07-05T10:00:36.161217+00:00"},{"alias_kind":"pith_short_8","alias_value":"F6SVHGEP","created_at":"2026-07-05T10:00:36.161217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.11917","citing_title":"Does VLM Classification Benefit from LLM Description Semantics?","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV","json":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV.json","graph_json":"https://pith.science/api/pith-number/F6SVHGEPFUGZ2USARYN4JM4YCV/graph.json","events_json":"https://pith.science/api/pith-number/F6SVHGEPFUGZ2USARYN4JM4YCV/events.json","paper":"https://pith.science/paper/F6SVHGEP"},"agent_actions":{"view_html":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV","download_json":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV.json","view_paper":"https://pith.science/paper/F6SVHGEP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12652&json=true","fetch_graph":"https://pith.science/api/pith-number/F6SVHGEPFUGZ2USARYN4JM4YCV/graph.json","fetch_events":"https://pith.science/api/pith-number/F6SVHGEPFUGZ2USARYN4JM4YCV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV/action/storage_attestation","attest_author":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV/action/author_attestation","sign_citation":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV/action/citation_signature","submit_replication":"https://pith.science/pith/F6SVHGEPFUGZ2USARYN4JM4YCV/action/replication_record"}},"created_at":"2026-07-05T10:00:36.161217+00:00","updated_at":"2026-07-05T10:00:36.161217+00:00"}