{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:4K7ADCNJY6IWYSXY2DCZT2DLMB","short_pith_number":"pith:4K7ADCNJ","schema_version":"1.0","canonical_sha256":"e2be0189a9c7916c4af8d0c599e86b606eedee136fc4d0be45c9c5b7a6f6eac1","source":{"kind":"arxiv","id":"2603.16100","version":2},"attestation_state":"computed","paper":{"title":"Reevaluating the Intra-Modal Misalignment Hypothesis in CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jonas Herzog, Yue Wang","submitted_at":"2026-03-17T03:49:55Z","abstract_excerpt":"Recent research suggested that the embeddings produced by CLIP-like contrastive language-image training are suboptimal for image-only tasks. The main theory is that the inter-modal (language-image) alignment loss ignores intra-modal (image-image) alignment, leading to poorly calibrated distances between images. In this study, we question this intra-modal misalignment hypothesis. We reexamine its foundational theoretical argument, the indicators used to support it, and the performance metrics affected. For the theoretical argument, we demonstrate that there are no such supposed degrees of freed"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.16100","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-03-17T03:49:55Z","cross_cats_sorted":[],"title_canon_sha256":"64fd9831cd69b607ed6abd1efc511b52f12ba8de5da4890d0b5510bedcc47c78","abstract_canon_sha256":"e783dd305a84d578eada4f93062ff4bae4c51dd9749e484960ca8d951b8e7c3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-26T02:04:08.856207Z","signature_b64":"0AUHc9ZnRcxcj/QwGHClid7U3CnpR2uwwpZs9EYQVYkywNJQGJJq7w/pHu/n27jEY6ck21FOCDBtfDyThMacAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2be0189a9c7916c4af8d0c599e86b606eedee136fc4d0be45c9c5b7a6f6eac1","last_reissued_at":"2026-05-26T02:04:08.855330Z","signature_status":"signed_v1","first_computed_at":"2026-05-26T02:04:08.855330Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reevaluating the Intra-Modal Misalignment Hypothesis in CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jonas Herzog, Yue Wang","submitted_at":"2026-03-17T03:49:55Z","abstract_excerpt":"Recent research suggested that the embeddings produced by CLIP-like contrastive language-image training are suboptimal for image-only tasks. The main theory is that the inter-modal (language-image) alignment loss ignores intra-modal (image-image) alignment, leading to poorly calibrated distances between images. In this study, we question this intra-modal misalignment hypothesis. We reexamine its foundational theoretical argument, the indicators used to support it, and the performance metrics affected. For the theoretical argument, we demonstrate that there are no such supposed degrees of freed"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.16100","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.16100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.16100","created_at":"2026-05-26T02:04:08.855463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.16100v2","created_at":"2026-05-26T02:04:08.855463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.16100","created_at":"2026-05-26T02:04:08.855463+00:00"},{"alias_kind":"pith_short_12","alias_value":"4K7ADCNJY6IW","created_at":"2026-05-26T02:04:08.855463+00:00"},{"alias_kind":"pith_short_16","alias_value":"4K7ADCNJY6IWYSXY","created_at":"2026-05-26T02:04:08.855463+00:00"},{"alias_kind":"pith_short_8","alias_value":"4K7ADCNJ","created_at":"2026-05-26T02:04:08.855463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB","json":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB.json","graph_json":"https://pith.science/api/pith-number/4K7ADCNJY6IWYSXY2DCZT2DLMB/graph.json","events_json":"https://pith.science/api/pith-number/4K7ADCNJY6IWYSXY2DCZT2DLMB/events.json","paper":"https://pith.science/paper/4K7ADCNJ"},"agent_actions":{"view_html":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB","download_json":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB.json","view_paper":"https://pith.science/paper/4K7ADCNJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.16100&json=true","fetch_graph":"https://pith.science/api/pith-number/4K7ADCNJY6IWYSXY2DCZT2DLMB/graph.json","fetch_events":"https://pith.science/api/pith-number/4K7ADCNJY6IWYSXY2DCZT2DLMB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB/action/storage_attestation","attest_author":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB/action/author_attestation","sign_citation":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB/action/citation_signature","submit_replication":"https://pith.science/pith/4K7ADCNJY6IWYSXY2DCZT2DLMB/action/replication_record"}},"created_at":"2026-05-26T02:04:08.855463+00:00","updated_at":"2026-05-26T02:04:08.855463+00:00"}