{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WDXRMHROZ7DGTERSFPE2I632LF","short_pith_number":"pith:WDXRMHRO","schema_version":"1.0","canonical_sha256":"b0ef161e2ecfc66992322bc9a47b7a5975906b733bf49f22b453163b5e4f5c5c","source":{"kind":"arxiv","id":"2303.15078","version":3},"attestation_state":"computed","paper":{"title":"Large Language Models are Diverse Role-Players for Summarization Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daxin Jiang, Linjun Shou, Ming Gong, Ning Wu, Shining Liang","submitted_at":"2023-03-27T10:40:59Z","abstract_excerpt":"Text summarization has a wide range of applications in many scenarios. The evaluation of the quality of the generated text is a complex problem. A big challenge to language evaluation is that there is a clear divergence between existing metrics and human evaluation. A document summary's quality can be assessed by human annotators on various criteria, both objective ones like grammar and correctness, and subjective ones like informativeness, succinctness, and appeal. Most of the automatic evaluation methods like BLUE/ROUGE may be not able to adequately capture the above dimensions. In this pape"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.15078","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-27T10:40:59Z","cross_cats_sorted":[],"title_canon_sha256":"4c4aa879ebecef8363746f77b31134e968879b786263e696ad02defc2d8e6c85","abstract_canon_sha256":"5346d4dca563a2361153ce16a0dbc45812c886688b75c4c19cfea9c791f0f782"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:52:03.261800Z","signature_b64":"4iFzCrQy5BVNHaWomu1D2A87fSFX5ppkUXuT5EhgP4X0/D3mDgo5MM0EOh3Z0LcDImKXFxXTFUkPSZpEPSoECw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0ef161e2ecfc66992322bc9a47b7a5975906b733bf49f22b453163b5e4f5c5c","last_reissued_at":"2026-07-05T06:52:03.261251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:52:03.261251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models are Diverse Role-Players for Summarization Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daxin Jiang, Linjun Shou, Ming Gong, Ning Wu, Shining Liang","submitted_at":"2023-03-27T10:40:59Z","abstract_excerpt":"Text summarization has a wide range of applications in many scenarios. The evaluation of the quality of the generated text is a complex problem. A big challenge to language evaluation is that there is a clear divergence between existing metrics and human evaluation. A document summary's quality can be assessed by human annotators on various criteria, both objective ones like grammar and correctness, and subjective ones like informativeness, succinctness, and appeal. Most of the automatic evaluation methods like BLUE/ROUGE may be not able to adequately capture the above dimensions. In this pape"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.15078","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.15078/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.15078","created_at":"2026-07-05T06:52:03.261323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.15078v3","created_at":"2026-07-05T06:52:03.261323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.15078","created_at":"2026-07-05T06:52:03.261323+00:00"},{"alias_kind":"pith_short_12","alias_value":"WDXRMHROZ7DG","created_at":"2026-07-05T06:52:03.261323+00:00"},{"alias_kind":"pith_short_16","alias_value":"WDXRMHROZ7DGTERS","created_at":"2026-07-05T06:52:03.261323+00:00"},{"alias_kind":"pith_short_8","alias_value":"WDXRMHRO","created_at":"2026-07-05T06:52:03.261323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2311.07911","citing_title":"Instruction-Following Evaluation for Large Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06608","citing_title":"The Prompt Report: A Systematic Survey of Prompt Engineering Techniques","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2308.07201","citing_title":"ChatEval: Towards Better LLM-based Evaluators through Multi-Agent Debate","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18234","citing_title":"Evaluating Multi-Hop Reasoning in RAG Systems: A Comparison of LLM-Based Retriever Evaluation Strategies","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF","json":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF.json","graph_json":"https://pith.science/api/pith-number/WDXRMHROZ7DGTERSFPE2I632LF/graph.json","events_json":"https://pith.science/api/pith-number/WDXRMHROZ7DGTERSFPE2I632LF/events.json","paper":"https://pith.science/paper/WDXRMHRO"},"agent_actions":{"view_html":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF","download_json":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF.json","view_paper":"https://pith.science/paper/WDXRMHRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.15078&json=true","fetch_graph":"https://pith.science/api/pith-number/WDXRMHROZ7DGTERSFPE2I632LF/graph.json","fetch_events":"https://pith.science/api/pith-number/WDXRMHROZ7DGTERSFPE2I632LF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF/action/storage_attestation","attest_author":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF/action/author_attestation","sign_citation":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF/action/citation_signature","submit_replication":"https://pith.science/pith/WDXRMHROZ7DGTERSFPE2I632LF/action/replication_record"}},"created_at":"2026-07-05T06:52:03.261323+00:00","updated_at":"2026-07-05T06:52:03.261323+00:00"}