{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G3M3HF4EOXVF5THPFLSAHG56FO","short_pith_number":"pith:G3M3HF4E","schema_version":"1.0","canonical_sha256":"36d9b3978475ea5eccef2ae4039bbe2ba42275957269cafd0bc343204f5547d7","source":{"kind":"arxiv","id":"2504.14891","version":1},"attestation_state":"computed","paper":{"title":"Retrieval Augmented Generation Evaluation in the Era of Large Language Models: A Comprehensive Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aoran Gan, Guoping Hu, Hao Yu, Kai Zhang, Qi Liu, Shiwei Tong, Wenyu Yan, Zhenya Huang","submitted_at":"2025-04-21T06:39:47Z","abstract_excerpt":"Recent advancements in Retrieval-Augmented Generation (RAG) have revolutionized natural language processing by integrating Large Language Models (LLMs) with external information retrieval, enabling accurate, up-to-date, and verifiable text generation across diverse applications. However, evaluating RAG systems presents unique challenges due to their hybrid architecture that combines retrieval and generation components, as well as their dependence on dynamic knowledge sources in the LLM era. In response, this paper provides a comprehensive survey of RAG evaluation methods and frameworks, system"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14891","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-21T06:39:47Z","cross_cats_sorted":[],"title_canon_sha256":"e77eb0c3a861e936ac1b3a6f86477b313188b21042bee1da9c84b82f40c41963","abstract_canon_sha256":"72b8884f022d0df0d92e39a6b344ff885b3924aa09964e4b3bf730506de8be86"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:52.363099Z","signature_b64":"CRl4S4Em0V6HudeiaEgUtmcYP1feGs3xpNfZ9yYk6XYQ06Gh0jc+PuWvN5Mij4xrKnJ2myr0lweHkxh8q9SyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36d9b3978475ea5eccef2ae4039bbe2ba42275957269cafd0bc343204f5547d7","last_reissued_at":"2026-07-05T10:51:52.362599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:52.362599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retrieval Augmented Generation Evaluation in the Era of Large Language Models: A Comprehensive Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aoran Gan, Guoping Hu, Hao Yu, Kai Zhang, Qi Liu, Shiwei Tong, Wenyu Yan, Zhenya Huang","submitted_at":"2025-04-21T06:39:47Z","abstract_excerpt":"Recent advancements in Retrieval-Augmented Generation (RAG) have revolutionized natural language processing by integrating Large Language Models (LLMs) with external information retrieval, enabling accurate, up-to-date, and verifiable text generation across diverse applications. However, evaluating RAG systems presents unique challenges due to their hybrid architecture that combines retrieval and generation components, as well as their dependence on dynamic knowledge sources in the LLM era. In response, this paper provides a comprehensive survey of RAG evaluation methods and frameworks, system"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14891","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14891/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14891","created_at":"2026-07-05T10:51:52.362662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14891v1","created_at":"2026-07-05T10:51:52.362662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14891","created_at":"2026-07-05T10:51:52.362662+00:00"},{"alias_kind":"pith_short_12","alias_value":"G3M3HF4EOXVF","created_at":"2026-07-05T10:51:52.362662+00:00"},{"alias_kind":"pith_short_16","alias_value":"G3M3HF4EOXVF5THP","created_at":"2026-07-05T10:51:52.362662+00:00"},{"alias_kind":"pith_short_8","alias_value":"G3M3HF4E","created_at":"2026-07-05T10:51:52.362662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.07523","citing_title":"Retrieval Augmented Generation Framework for the Nepali Legal Domain Question Answering","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05901","citing_title":"Reducing Hallucinations in Complex Question Answering using Simple Graph-based Retrieval-Augmented Generation (long version)","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29640","citing_title":"VikingMem: A Memory Base Management System for Stateful LLM-based Applications","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12138","citing_title":"Retrieval-Augmented Generation Must Move Beyond Factual Grounding to Represent Diverse Opinions","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12047","citing_title":"Empirical Evaluation of PDF Parsing and Chunking for Financial Question Answering with RAG","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO","json":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO.json","graph_json":"https://pith.science/api/pith-number/G3M3HF4EOXVF5THPFLSAHG56FO/graph.json","events_json":"https://pith.science/api/pith-number/G3M3HF4EOXVF5THPFLSAHG56FO/events.json","paper":"https://pith.science/paper/G3M3HF4E"},"agent_actions":{"view_html":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO","download_json":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO.json","view_paper":"https://pith.science/paper/G3M3HF4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14891&json=true","fetch_graph":"https://pith.science/api/pith-number/G3M3HF4EOXVF5THPFLSAHG56FO/graph.json","fetch_events":"https://pith.science/api/pith-number/G3M3HF4EOXVF5THPFLSAHG56FO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO/action/storage_attestation","attest_author":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO/action/author_attestation","sign_citation":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO/action/citation_signature","submit_replication":"https://pith.science/pith/G3M3HF4EOXVF5THPFLSAHG56FO/action/replication_record"}},"created_at":"2026-07-05T10:51:52.362662+00:00","updated_at":"2026-07-05T10:51:52.362662+00:00"}