{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LFDMS6DZCEHHE2SOPVSGK2EFNB","short_pith_number":"pith:LFDMS6DZ","schema_version":"1.0","canonical_sha256":"5946c97879110e726a4e7d64656885684531513132bfbd7c946f5a8b6c0b6a08","source":{"kind":"arxiv","id":"2503.17782","version":2},"attestation_state":"computed","paper":{"title":"GOAL: Global-local Object Alignment Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chanho Eom, Hyungyu Choi, Young Kyun Jang","submitted_at":"2025-03-22T14:27:32Z","abstract_excerpt":"Vision-language models like CLIP have shown impressive capabilities in aligning images and text, but they often struggle with lengthy and detailed text descriptions because of their training focus on short and concise captions. We present GOAL (Global-local Object Alignment Learning), a novel fine-tuning method that enhances CLIP's ability to handle lengthy text by leveraging both global and local semantic alignments between image and lengthy text. Our approach consists of two key components: Local Image-Sentence Matching (LISM), which identifies corresponding pairs between image segments and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.17782","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-22T14:27:32Z","cross_cats_sorted":[],"title_canon_sha256":"f7ff44c6f595779341f950731de3b502aec42350226c4b2726826acd2801bef3","abstract_canon_sha256":"4bb3a21db77293c6e9bdcbfd97449bf91b2556fc4dfdefb25851f14565cd7779"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:45.685897Z","signature_b64":"HwVJnHwXDpLQEJrgO0+PsG1L+01G3Q2kixPpi+uvSoUZFD6UoM//YVsEohlIOiAOkbsO+LIkMnDauWIJeh1aAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5946c97879110e726a4e7d64656885684531513132bfbd7c946f5a8b6c0b6a08","last_reissued_at":"2026-07-05T10:38:45.685405Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:45.685405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GOAL: Global-local Object Alignment Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chanho Eom, Hyungyu Choi, Young Kyun Jang","submitted_at":"2025-03-22T14:27:32Z","abstract_excerpt":"Vision-language models like CLIP have shown impressive capabilities in aligning images and text, but they often struggle with lengthy and detailed text descriptions because of their training focus on short and concise captions. We present GOAL (Global-local Object Alignment Learning), a novel fine-tuning method that enhances CLIP's ability to handle lengthy text by leveraging both global and local semantic alignments between image and lengthy text. Our approach consists of two key components: Local Image-Sentence Matching (LISM), which identifies corresponding pairs between image segments and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.17782","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.17782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.17782","created_at":"2026-07-05T10:38:45.685465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.17782v2","created_at":"2026-07-05T10:38:45.685465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.17782","created_at":"2026-07-05T10:38:45.685465+00:00"},{"alias_kind":"pith_short_12","alias_value":"LFDMS6DZCEHH","created_at":"2026-07-05T10:38:45.685465+00:00"},{"alias_kind":"pith_short_16","alias_value":"LFDMS6DZCEHHE2SO","created_at":"2026-07-05T10:38:45.685465+00:00"},{"alias_kind":"pith_short_8","alias_value":"LFDMS6DZ","created_at":"2026-07-05T10:38:45.685465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00905","citing_title":"Spotlighter: Revisiting Prompt Tuning from a Representative Mining View","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB","json":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB.json","graph_json":"https://pith.science/api/pith-number/LFDMS6DZCEHHE2SOPVSGK2EFNB/graph.json","events_json":"https://pith.science/api/pith-number/LFDMS6DZCEHHE2SOPVSGK2EFNB/events.json","paper":"https://pith.science/paper/LFDMS6DZ"},"agent_actions":{"view_html":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB","download_json":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB.json","view_paper":"https://pith.science/paper/LFDMS6DZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.17782&json=true","fetch_graph":"https://pith.science/api/pith-number/LFDMS6DZCEHHE2SOPVSGK2EFNB/graph.json","fetch_events":"https://pith.science/api/pith-number/LFDMS6DZCEHHE2SOPVSGK2EFNB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB/action/storage_attestation","attest_author":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB/action/author_attestation","sign_citation":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB/action/citation_signature","submit_replication":"https://pith.science/pith/LFDMS6DZCEHHE2SOPVSGK2EFNB/action/replication_record"}},"created_at":"2026-07-05T10:38:45.685465+00:00","updated_at":"2026-07-05T10:38:45.685465+00:00"}