{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5VE26SN77STVIL7WWWC6U3BYT3","short_pith_number":"pith:5VE26SN7","schema_version":"1.0","canonical_sha256":"ed49af49bffca7542ff6b585ea6c389ee6f0b32bba1fcabe1d7511e12cc5c6ce","source":{"kind":"arxiv","id":"2111.02110","version":3},"attestation_state":"computed","paper":{"title":"Automatic Evaluation and Moderation of Open-domain Dialogue Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Alexander Rudnicky, Chen Zhang, Jo\\~ao Sedoc, Luis Fernando D'Haro, Rafael Banchs","submitted_at":"2021-11-03T10:08:05Z","abstract_excerpt":"The development of Open-Domain Dialogue Systems (ODS)is a trending topic due to the large number of research challenges, large societal and business impact, and advances in the underlying technology. However, the development of these kinds of systems requires two important characteristics:1) automatic evaluation mechanisms that show high correlations with human judgements across multiple dialogue evaluation aspects (with explainable features for providing constructive and explicit feedback on the quality of generative models' responses for quick development and deployment)and 2) mechanisms tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.02110","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-11-03T10:08:05Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"c635aa4ad4b079cf51401df27288424e9c0e177d0bb149837c3cfaa020d3e0c8","abstract_canon_sha256":"f3529acb35b164b1cea9830ba0f1ed3b259d80d96da40d85fc0a6183691cd305"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:43:45.181395Z","signature_b64":"K8f+SMJ3zj3fEiCqVM7/zGSDGoX7PMH9RlfqQGr2fKTus90OyZYene7yl+tlMiTZ0eryMdz/4fA3DylwcTvmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed49af49bffca7542ff6b585ea6c389ee6f0b32bba1fcabe1d7511e12cc5c6ce","last_reissued_at":"2026-07-05T03:43:45.180917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:43:45.180917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatic Evaluation and Moderation of Open-domain Dialogue Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Alexander Rudnicky, Chen Zhang, Jo\\~ao Sedoc, Luis Fernando D'Haro, Rafael Banchs","submitted_at":"2021-11-03T10:08:05Z","abstract_excerpt":"The development of Open-Domain Dialogue Systems (ODS)is a trending topic due to the large number of research challenges, large societal and business impact, and advances in the underlying technology. However, the development of these kinds of systems requires two important characteristics:1) automatic evaluation mechanisms that show high correlations with human judgements across multiple dialogue evaluation aspects (with explainable features for providing constructive and explicit feedback on the quality of generative models' responses for quick development and deployment)and 2) mechanisms tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.02110","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.02110/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.02110","created_at":"2026-07-05T03:43:45.180975+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.02110v3","created_at":"2026-07-05T03:43:45.180975+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.02110","created_at":"2026-07-05T03:43:45.180975+00:00"},{"alias_kind":"pith_short_12","alias_value":"5VE26SN77STV","created_at":"2026-07-05T03:43:45.180975+00:00"},{"alias_kind":"pith_short_16","alias_value":"5VE26SN77STVIL7W","created_at":"2026-07-05T03:43:45.180975+00:00"},{"alias_kind":"pith_short_8","alias_value":"5VE26SN7","created_at":"2026-07-05T03:43:45.180975+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":290,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3","json":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3.json","graph_json":"https://pith.science/api/pith-number/5VE26SN77STVIL7WWWC6U3BYT3/graph.json","events_json":"https://pith.science/api/pith-number/5VE26SN77STVIL7WWWC6U3BYT3/events.json","paper":"https://pith.science/paper/5VE26SN7"},"agent_actions":{"view_html":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3","download_json":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3.json","view_paper":"https://pith.science/paper/5VE26SN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.02110&json=true","fetch_graph":"https://pith.science/api/pith-number/5VE26SN77STVIL7WWWC6U3BYT3/graph.json","fetch_events":"https://pith.science/api/pith-number/5VE26SN77STVIL7WWWC6U3BYT3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3/action/storage_attestation","attest_author":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3/action/author_attestation","sign_citation":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3/action/citation_signature","submit_replication":"https://pith.science/pith/5VE26SN77STVIL7WWWC6U3BYT3/action/replication_record"}},"created_at":"2026-07-05T03:43:45.180975+00:00","updated_at":"2026-07-05T03:43:45.180975+00:00"}