{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IHM6STAKKXIQ4HANKFSEPCMV6U","short_pith_number":"pith:IHM6STAK","schema_version":"1.0","canonical_sha256":"41d9e94c0a55d10e1c0d5164478995f537a5733d448dcfd4eeb3e0e145c5dc03","source":{"kind":"arxiv","id":"2305.13693","version":1},"attestation_state":"computed","paper":{"title":"Automated Metrics for Medical Multi-Document Summarization Disagree with Human Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bailey E. Kuehl, Byron C. Wallace, Erin Bransom, Jay DeYoung, Lucy Lu Wang, Thinh Hung Truong, Yulia Otmakhova","submitted_at":"2023-05-23T05:00:59Z","abstract_excerpt":"Evaluating multi-document summarization (MDS) quality is difficult. This is especially true in the case of MDS for biomedical literature reviews, where models must synthesize contradicting evidence reported across different documents. Prior work has shown that rather than performing the task, models may exploit shortcuts that are difficult to detect using standard n-gram similarity metrics such as ROUGE. Better automated evaluation metrics are needed, but few resources exist to assess metrics when they are proposed. Therefore, we introduce a dataset of human-assessed summary quality facets and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13693","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T05:00:59Z","cross_cats_sorted":[],"title_canon_sha256":"968b60c71ba541b6bbb86182d17c85f3858d28e9f9d3150456c1734389a1fb51","abstract_canon_sha256":"09e2326fc79fe6c873e514342048217f23ff699306a2b39bd68ab8ff735cb779"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:56.173204Z","signature_b64":"4Q6u6vQRDucG1mCNJizQ+hr7WF4oKoZzdPbQL1XmojZNsV00TfVGuqeEoPtwk8XNWtLlCTXcmjn1Cvfz8AZoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41d9e94c0a55d10e1c0d5164478995f537a5733d448dcfd4eeb3e0e145c5dc03","last_reissued_at":"2026-07-05T06:12:56.172874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:56.172874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automated Metrics for Medical Multi-Document Summarization Disagree with Human Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bailey E. Kuehl, Byron C. Wallace, Erin Bransom, Jay DeYoung, Lucy Lu Wang, Thinh Hung Truong, Yulia Otmakhova","submitted_at":"2023-05-23T05:00:59Z","abstract_excerpt":"Evaluating multi-document summarization (MDS) quality is difficult. This is especially true in the case of MDS for biomedical literature reviews, where models must synthesize contradicting evidence reported across different documents. Prior work has shown that rather than performing the task, models may exploit shortcuts that are difficult to detect using standard n-gram similarity metrics such as ROUGE. Better automated evaluation metrics are needed, but few resources exist to assess metrics when they are proposed. Therefore, we introduce a dataset of human-assessed summary quality facets and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13693","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13693/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13693","created_at":"2026-07-05T06:12:56.172928+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13693v1","created_at":"2026-07-05T06:12:56.172928+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13693","created_at":"2026-07-05T06:12:56.172928+00:00"},{"alias_kind":"pith_short_12","alias_value":"IHM6STAKKXIQ","created_at":"2026-07-05T06:12:56.172928+00:00"},{"alias_kind":"pith_short_16","alias_value":"IHM6STAKKXIQ4HAN","created_at":"2026-07-05T06:12:56.172928+00:00"},{"alias_kind":"pith_short_8","alias_value":"IHM6STAK","created_at":"2026-07-05T06:12:56.172928+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07936","citing_title":"Illusions of the Gold Standard: A Large-scale Analysis of Human Evaluation Protocols for Long-form Text Generation","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U","json":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U.json","graph_json":"https://pith.science/api/pith-number/IHM6STAKKXIQ4HANKFSEPCMV6U/graph.json","events_json":"https://pith.science/api/pith-number/IHM6STAKKXIQ4HANKFSEPCMV6U/events.json","paper":"https://pith.science/paper/IHM6STAK"},"agent_actions":{"view_html":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U","download_json":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U.json","view_paper":"https://pith.science/paper/IHM6STAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13693&json=true","fetch_graph":"https://pith.science/api/pith-number/IHM6STAKKXIQ4HANKFSEPCMV6U/graph.json","fetch_events":"https://pith.science/api/pith-number/IHM6STAKKXIQ4HANKFSEPCMV6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U/action/storage_attestation","attest_author":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U/action/author_attestation","sign_citation":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U/action/citation_signature","submit_replication":"https://pith.science/pith/IHM6STAKKXIQ4HANKFSEPCMV6U/action/replication_record"}},"created_at":"2026-07-05T06:12:56.172928+00:00","updated_at":"2026-07-05T06:12:56.172928+00:00"}