{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CGKKVPBLX6S37SYY7W5WFBNVBV","short_pith_number":"pith:CGKKVPBL","schema_version":"1.0","canonical_sha256":"1194aabc2bbfa5bfcb18fdbb6285b50d59fba7330884d6e03713ab8cbeaee838","source":{"kind":"arxiv","id":"2402.11398","version":2},"attestation_state":"computed","paper":{"title":"Reasoning before Comparison: LLM-Enhanced Semantic Similarity Metrics for Domain Specialized Text Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andrea Sikora, Huaqin Zhao, Peng Shu, Shaochen Xu, Sheng Li, Tianming Liu, Wenxiong Liao, Xiang Li, Zhengliang Liu, Zihao Wu","submitted_at":"2024-02-17T22:46:44Z","abstract_excerpt":"In this study, we leverage LLM to enhance the semantic analysis and develop similarity metrics for texts, addressing the limitations of traditional unsupervised NLP metrics like ROUGE and BLEU. We develop a framework where LLMs such as GPT-4 are employed for zero-shot text identification and label generation for radiology reports, where the labels are then used as measurements for text similarity. By testing the proposed framework on the MIMIC data, we find that GPT-4 generated labels can significantly improve the semantic similarity assessment, with scores more closely aligned with clinical g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11398","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-17T22:46:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e01991efe135402505fb02396752934c9b6bdd667944816709199f0fcc10d9d1","abstract_canon_sha256":"527252a20116c4c79ba41533af01a924a3bf2d7202911b8269585b5a14fc1c95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:29.122168Z","signature_b64":"xfSsHEiOmazk2QVGNuSF71zpBqQuMMyx6aCjeJB4Ur5wQQuFjQVFATicVf0KVZLMvK5CWPC8cwEyyJRS3Zn8CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1194aabc2bbfa5bfcb18fdbb6285b50d59fba7330884d6e03713ab8cbeaee838","last_reissued_at":"2026-07-05T07:47:29.121648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:29.121648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning before Comparison: LLM-Enhanced Semantic Similarity Metrics for Domain Specialized Text Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andrea Sikora, Huaqin Zhao, Peng Shu, Shaochen Xu, Sheng Li, Tianming Liu, Wenxiong Liao, Xiang Li, Zhengliang Liu, Zihao Wu","submitted_at":"2024-02-17T22:46:44Z","abstract_excerpt":"In this study, we leverage LLM to enhance the semantic analysis and develop similarity metrics for texts, addressing the limitations of traditional unsupervised NLP metrics like ROUGE and BLEU. We develop a framework where LLMs such as GPT-4 are employed for zero-shot text identification and label generation for radiology reports, where the labels are then used as measurements for text similarity. By testing the proposed framework on the MIMIC data, we find that GPT-4 generated labels can significantly improve the semantic similarity assessment, with scores more closely aligned with clinical g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11398","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11398","created_at":"2026-07-05T07:47:29.121715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11398v2","created_at":"2026-07-05T07:47:29.121715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11398","created_at":"2026-07-05T07:47:29.121715+00:00"},{"alias_kind":"pith_short_12","alias_value":"CGKKVPBLX6S3","created_at":"2026-07-05T07:47:29.121715+00:00"},{"alias_kind":"pith_short_16","alias_value":"CGKKVPBLX6S37SYY","created_at":"2026-07-05T07:47:29.121715+00:00"},{"alias_kind":"pith_short_8","alias_value":"CGKKVPBL","created_at":"2026-07-05T07:47:29.121715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.06209","citing_title":"TelcoAgent-Bench: A Multilingual Benchmark for Telecom AI Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":146,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV","json":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV.json","graph_json":"https://pith.science/api/pith-number/CGKKVPBLX6S37SYY7W5WFBNVBV/graph.json","events_json":"https://pith.science/api/pith-number/CGKKVPBLX6S37SYY7W5WFBNVBV/events.json","paper":"https://pith.science/paper/CGKKVPBL"},"agent_actions":{"view_html":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV","download_json":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV.json","view_paper":"https://pith.science/paper/CGKKVPBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11398&json=true","fetch_graph":"https://pith.science/api/pith-number/CGKKVPBLX6S37SYY7W5WFBNVBV/graph.json","fetch_events":"https://pith.science/api/pith-number/CGKKVPBLX6S37SYY7W5WFBNVBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV/action/storage_attestation","attest_author":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV/action/author_attestation","sign_citation":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV/action/citation_signature","submit_replication":"https://pith.science/pith/CGKKVPBLX6S37SYY7W5WFBNVBV/action/replication_record"}},"created_at":"2026-07-05T07:47:29.121715+00:00","updated_at":"2026-07-05T07:47:29.121715+00:00"}