{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:4PQYXXNT3XJLFDYBVQC2WNCDH6","short_pith_number":"pith:4PQYXXNT","canonical_record":{"source":{"id":"2211.09110","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T18:51:34Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8b61ecb21f40f5050d219aa6879d32a1f08e0e2a55ff5e7ad0fa2e141ccbc56c","abstract_canon_sha256":"5e8efe20a353e260589415162a56e5c0872f9e62d4f58f9e1aa97f12c642bd32"},"schema_version":"1.0"},"canonical_sha256":"e3e18bddb3ddd2b28f01ac05ab34433faca427f4e0532cbe6708657207dca654","source":{"kind":"arxiv","id":"2211.09110","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2211.09110","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"arxiv_version","alias_value":"2211.09110v2","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09110","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_12","alias_value":"4PQYXXNT3XJL","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_16","alias_value":"4PQYXXNT3XJLFDYB","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_8","alias_value":"4PQYXXNT","created_at":"2026-07-05T06:55:50Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:4PQYXXNT3XJLFDYBVQC2WNCDH6","target":"record","payload":{"canonical_record":{"source":{"id":"2211.09110","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T18:51:34Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8b61ecb21f40f5050d219aa6879d32a1f08e0e2a55ff5e7ad0fa2e141ccbc56c","abstract_canon_sha256":"5e8efe20a353e260589415162a56e5c0872f9e62d4f58f9e1aa97f12c642bd32"},"schema_version":"1.0"},"canonical_sha256":"e3e18bddb3ddd2b28f01ac05ab34433faca427f4e0532cbe6708657207dca654","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:50.810673Z","signature_b64":"0S+8V/4T0dxSNxOqNMQpkNmJUpTEorOjsR0X/BFyqmzVvw/jwfovK0470Vbvf207nSpvrou+9ozZKhapENfKAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3e18bddb3ddd2b28f01ac05ab34433faca427f4e0532cbe6708657207dca654","last_reissued_at":"2026-07-05T06:55:50.810187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:50.810187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2211.09110","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:55:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QPTRdj7VBnUCQkhLfw6npL1lfbD/lRHBM/NlSU/fdUSzgOpvVM+4inEtnVayBlRv2ORUn3Wt8iz1V8u7dmcyCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T05:07:01.728467Z"},"content_sha256":"ebdc9e87c3d6476eb04a6ffb198164f892e0db2497bc32469482a81626d9a882","schema_version":"1.0","event_id":"sha256:ebdc9e87c3d6476eb04a6ffb198164f892e0db2497bc32469482a81626d9a882"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:4PQYXXNT3XJLFDYBVQC2WNCDH6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Holistic Evaluation of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ananya Kumar, Benjamin Newman, Binhang Yuan, Bobby Yan, Ce Zhang, Christian Cosgrove, Christopher D. Manning, Christopher R\\'e, Deepak Narayanan, Diana Acosta-Navas, Dilara Soylu, Dimitris Tsipras, Drew A. Hudson, Eric Zelikman, Esin Durmus, Faisal Ladhak, Frieda Rong, Hongyu Ren, Huaxiu Yao, Jue Wang, Keshav Santhanam, Laurel Orr, Lucia Zheng, Mert Yuksekgonul, Michihiro Yasunaga, Mirac Suzgun, Nathan Kim, Neel Guha, Niladri Chatterji, Omar Khattab, Percy Liang, Peter Henderson, Qian Huang, Rishi Bommasani, Ryan Chi, Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Tony Lee, Vishrav Chaudhary, William Wang, Xuechen Li, Yian Zhang, Yifan Mai, Yuhuai Wu, Yuhui Zhang, Yuta Koreeda","submitted_at":"2022-11-16T18:51:34Z","abstract_excerpt":"Language models (LMs) are becoming the foundation for almost all major language technologies, but their capabilities, limitations, and risks are not well understood. We present Holistic Evaluation of Language Models (HELM) to improve the transparency of language models. First, we taxonomize the vast space of potential scenarios (i.e. use cases) and metrics (i.e. desiderata) that are of interest for LMs. Then we select a broad subset based on coverage and feasibility, noting what's missing or underrepresented (e.g. question answering for neglected English dialects, metrics for trustworthiness)."},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We improve this to 96.0%: now all 30 models have been densely benchmarked on the same core scenarios and metrics under standardized conditions. Our evaluation surfaces 25 top-level findings.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The selection of a broad but feasible subset of scenarios and metrics from the full taxonomy is sufficient to deliver a holistic view, even while the paper explicitly notes missing or underrepresented areas such as question answering for neglected English dialects and metrics for trustworthiness.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"HELM establishes a multi-metric evaluation covering 30 language models on 42 scenarios (16 core) to raise average scenario coverage from 17.9% to 96% under uniform conditions while releasing all prompts, completions, and a toolkit.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"6a70928f421528a515b1171351ce455e739868126c73acce62221e18b8b435da"},"source":{"id":"2211.09110","kind":"arxiv","version":2},"verdict":{"id":"f9f25deb-e0b1-4da7-be48-d5403e3b1731","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-24T10:04:27.979965Z","strongest_claim":"We improve this to 96.0%: now all 30 models have been densely benchmarked on the same core scenarios and metrics under standardized conditions. Our evaluation surfaces 25 top-level findings.","one_line_summary":"HELM establishes a multi-metric evaluation covering 30 language models on 42 scenarios (16 core) to raise average scenario coverage from 17.9% to 96% under uniform conditions while releasing all prompts, completions, and a toolkit.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The selection of a broad but feasible subset of scenarios and metrics from the full taxonomy is sufficient to deliver a holistic view, even while the paper explicitly notes missing or underrepresented areas such as question answering for neglected English dialects and metrics for trustworthiness.","pith_extraction_headline":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.09110/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":21,"sample":[{"doi":"10.18653/v1/2021.naacl-main.385","year":2021,"title":"Language Models are Few-Shot Learners","work_id":"214732c0-2edd-44a0-af9e-28184a2b8279","ref_index":1,"cited_arxiv_id":"2005.14165","is_internal_anchor":true},{"doi":"10.18653/v1/2021.acl-long.150","year":2021,"title":"doi: 10.18653/v1/2021.acl-long.150","work_id":"28ca0026-3906-4281-9006-088b556137d2","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.5281/zenodo.4761960","year":2018,"title":"URLhttps://glottolog.org/accessed2021-08-08","work_id":"b03a94d5-21f1-4c7f-8e70-54e70f8cd4db","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.18653/v1/2021.eacl-main.225","year":2021,"title":"Measuring Coding Challenge Competence With APPS","work_id":"c014c12f-1080-4cb2-ae03-ab6b7c09445c","ref_index":4,"cited_arxiv_id":"2105.09938","is_internal_anchor":true},{"doi":"10.1093/oxfordhb/9780199286546.001.0001/","year":2021,"title":"In Christopher Hitchcock & Alan Hajek, edi- tors: Oxford Handbook of Probability and Philosophy , Oxford University Press, pp","work_id":"14395009-699b-40c7-94de-6004ef131037","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":21,"snapshot_sha256":"acf92c8ccb24de906fa513991324cadf28c42956427d1f31fc748d008597ba99","internal_anchors":7},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"f9f25deb-e0b1-4da7-be48-d5403e3b1731"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:55:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1AnE3REF9AQMoi++PKje0QJwnLzPLeeo5r6qvSeJD24w58rPYJzior8XSXzK/KQEb8QtRJifqmZTwffo6RKODg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T05:07:01.729800Z"},"content_sha256":"cdff8494161b699674a682ce76eb63e8c79349d0190c87eefcbc24521f1a8773","schema_version":"1.0","event_id":"sha256:cdff8494161b699674a682ce76eb63e8c79349d0190c87eefcbc24521f1a8773"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/bundle.json","state_url":"https://pith.science/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T05:07:01Z","links":{"resolver":"https://pith.science/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6","bundle":"https://pith.science/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/bundle.json","state":"https://pith.science/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4PQYXXNT3XJLFDYBVQC2WNCDH6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:4PQYXXNT3XJLFDYBVQC2WNCDH6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5e8efe20a353e260589415162a56e5c0872f9e62d4f58f9e1aa97f12c642bd32","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T18:51:34Z","title_canon_sha256":"8b61ecb21f40f5050d219aa6879d32a1f08e0e2a55ff5e7ad0fa2e141ccbc56c"},"schema_version":"1.0","source":{"id":"2211.09110","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2211.09110","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"arxiv_version","alias_value":"2211.09110v2","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09110","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_12","alias_value":"4PQYXXNT3XJL","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_16","alias_value":"4PQYXXNT3XJLFDYB","created_at":"2026-07-05T06:55:50Z"},{"alias_kind":"pith_short_8","alias_value":"4PQYXXNT","created_at":"2026-07-05T06:55:50Z"}],"graph_snapshots":[{"event_id":"sha256:cdff8494161b699674a682ce76eb63e8c79349d0190c87eefcbc24521f1a8773","target":"graph","created_at":"2026-07-05T06:55:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We improve this to 96.0%: now all 30 models have been densely benchmarked on the same core scenarios and metrics under standardized conditions. Our evaluation surfaces 25 top-level findings."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The selection of a broad but feasible subset of scenarios and metrics from the full taxonomy is sufficient to deliver a holistic view, even while the paper explicitly notes missing or underrepresented areas such as question answering for neglected English dialects and metrics for trustworthiness."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"HELM establishes a multi-metric evaluation covering 30 language models on 42 scenarios (16 core) to raise average scenario coverage from 17.9% to 96% under uniform conditions while releasing all prompts, completions, and a toolkit."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated."}],"snapshot_sha256":"6a70928f421528a515b1171351ce455e739868126c73acce62221e18b8b435da"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2211.09110/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language models (LMs) are becoming the foundation for almost all major language technologies, but their capabilities, limitations, and risks are not well understood. We present Holistic Evaluation of Language Models (HELM) to improve the transparency of language models. First, we taxonomize the vast space of potential scenarios (i.e. use cases) and metrics (i.e. desiderata) that are of interest for LMs. Then we select a broad subset based on coverage and feasibility, noting what's missing or underrepresented (e.g. question answering for neglected English dialects, metrics for trustworthiness).","authors_text":"Ananya Kumar, Benjamin Newman, Binhang Yuan, Bobby Yan, Ce Zhang, Christian Cosgrove, Christopher D. Manning, Christopher R\\'e, Deepak Narayanan, Diana Acosta-Navas, Dilara Soylu, Dimitris Tsipras, Drew A. Hudson, Eric Zelikman, Esin Durmus, Faisal Ladhak, Frieda Rong, Hongyu Ren, Huaxiu Yao, Jue Wang, Keshav Santhanam, Laurel Orr, Lucia Zheng, Mert Yuksekgonul, Michihiro Yasunaga, Mirac Suzgun, Nathan Kim, Neel Guha, Niladri Chatterji, Omar Khattab, Percy Liang, Peter Henderson, Qian Huang, Rishi Bommasani, Ryan Chi, Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Tony Lee, Vishrav Chaudhary, William Wang, Xuechen Li, Yian Zhang, Yifan Mai, Yuhuai Wu, Yuhui Zhang, Yuta Koreeda","cross_cats":["cs.AI","cs.LG"],"headline":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models"},"references":{"count":21,"internal_anchors":7,"resolved_work":21,"sample":[{"cited_arxiv_id":"2005.14165","doi":"10.18653/v1/2021.naacl-main.385","is_internal_anchor":true,"ref_index":1,"title":"Language Models are Few-Shot Learners","work_id":"214732c0-2edd-44a0-af9e-28184a2b8279","year":2021},{"cited_arxiv_id":"","doi":"10.18653/v1/2021.acl-long.150","is_internal_anchor":false,"ref_index":2,"title":"doi: 10.18653/v1/2021.acl-long.150","work_id":"28ca0026-3906-4281-9006-088b556137d2","year":2021},{"cited_arxiv_id":"","doi":"10.5281/zenodo.4761960","is_internal_anchor":false,"ref_index":3,"title":"URLhttps://glottolog.org/accessed2021-08-08","work_id":"b03a94d5-21f1-4c7f-8e70-54e70f8cd4db","year":2018},{"cited_arxiv_id":"2105.09938","doi":"10.18653/v1/2021.eacl-main.225","is_internal_anchor":true,"ref_index":4,"title":"Measuring Coding Challenge Competence With APPS","work_id":"c014c12f-1080-4cb2-ae03-ab6b7c09445c","year":2021},{"cited_arxiv_id":"","doi":"10.1093/oxfordhb/9780199286546.001.0001/","is_internal_anchor":false,"ref_index":5,"title":"In Christopher Hitchcock & Alan Hajek, edi- tors: Oxford Handbook of Probability and Philosophy , Oxford University Press, pp","work_id":"14395009-699b-40c7-94de-6004ef131037","year":2021}],"snapshot_sha256":"acf92c8ccb24de906fa513991324cadf28c42956427d1f31fc748d008597ba99"},"source":{"id":"2211.09110","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-24T10:04:27.979965Z","id":"f9f25deb-e0b1-4da7-be48-d5403e3b1731","model_set":{"reader":"grok-4.3"},"one_line_summary":"HELM establishes a multi-metric evaluation covering 30 language models on 42 scenarios (16 core) to raise average scenario coverage from 17.9% to 96% under uniform conditions while releasing all prompts, completions, and a toolkit.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Language models are now densely benchmarked on the same 42 scenarios and 7 metrics under standardized conditions for all 30 models evaluated.","strongest_claim":"We improve this to 96.0%: now all 30 models have been densely benchmarked on the same core scenarios and metrics under standardized conditions. Our evaluation surfaces 25 top-level findings.","weakest_assumption":"The selection of a broad but feasible subset of scenarios and metrics from the full taxonomy is sufficient to deliver a holistic view, even while the paper explicitly notes missing or underrepresented areas such as question answering for neglected English dialects and metrics for trustworthiness."}},"verdict_id":"f9f25deb-e0b1-4da7-be48-d5403e3b1731"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ebdc9e87c3d6476eb04a6ffb198164f892e0db2497bc32469482a81626d9a882","target":"record","created_at":"2026-07-05T06:55:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5e8efe20a353e260589415162a56e5c0872f9e62d4f58f9e1aa97f12c642bd32","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T18:51:34Z","title_canon_sha256":"8b61ecb21f40f5050d219aa6879d32a1f08e0e2a55ff5e7ad0fa2e141ccbc56c"},"schema_version":"1.0","source":{"id":"2211.09110","kind":"arxiv","version":2}},"canonical_sha256":"e3e18bddb3ddd2b28f01ac05ab34433faca427f4e0532cbe6708657207dca654","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e3e18bddb3ddd2b28f01ac05ab34433faca427f4e0532cbe6708657207dca654","first_computed_at":"2026-07-05T06:55:50.810187Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:55:50.810187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"0S+8V/4T0dxSNxOqNMQpkNmJUpTEorOjsR0X/BFyqmzVvw/jwfovK0470Vbvf207nSpvrou+9ozZKhapENfKAw==","signature_status":"signed_v1","signed_at":"2026-07-05T06:55:50.810673Z","signed_message":"canonical_sha256_bytes"},"source_id":"2211.09110","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ebdc9e87c3d6476eb04a6ffb198164f892e0db2497bc32469482a81626d9a882","sha256:cdff8494161b699674a682ce76eb63e8c79349d0190c87eefcbc24521f1a8773"],"state_sha256":"7e80caf404f655621507c21883ffb16e879b3373271db6a5f378a8f23404f7bb"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UNN+zmOaz0OHmw1Cn0AJV2Y+5TUPWp2RTJQDdJ2Z4PWJMRgFc0CFzemCu5o/KByrww5PnKIqR0pAv4p56ylEAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T05:07:01.739130Z","bundle_sha256":"1e156bec8059a8fb211f2608d0514011de0e988d595dc2008f77b5095de5954f"}}