{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3SMX3ST62SRWMZ6BVSJN6ZAS5Z","short_pith_number":"pith:3SMX3ST6","schema_version":"1.0","canonical_sha256":"dc997dca7ed4a36667c1ac92df6412ee61a665eb4b6b8060eef74f828f8cb206","source":{"kind":"arxiv","id":"2309.17012","version":3},"attestation_state":"computed","paper":{"title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dongyeop Kang, Jong Inn Park, Minhwa Lee, Ryan Koo, Vipul Raheja, Zae Myung Kim","submitted_at":"2023-09-29T06:53:10Z","abstract_excerpt":"Large Language Models are cognitively biased judges. Large Language Models (LLMs) have recently been shown to be effective as automatic evaluators with simple prompting and in-context learning. In this work, we assemble 15 LLMs of four different size ranges and evaluate their output responses by preference ranking from the other LLMs as evaluators, such as System Star is better than System Square. We then evaluate the quality of ranking outputs introducing the Cognitive Bias Benchmark for LLMs as Evaluators (CoBBLEr), a benchmark to measure six different cognitive biases in LLM evaluation outp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.17012","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-29T06:53:10Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a620ddcc5fa85143441ccb17663e8c56943ba1b0489f201bca618916b1b21cfb","abstract_canon_sha256":"a034799f8ecb228ca9409220e0ab867fb849a14b2e0cf5a70294dc1064824e9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:11:22.269496Z","signature_b64":"XWHglaydpPh6EWpJ7yiMzGj0Telewv71SwHaTLAP75p7p5KaoCqjKT3brSMdTYBxAkS7KlhZT7lUILnQ6a1PCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc997dca7ed4a36667c1ac92df6412ee61a665eb4b6b8060eef74f828f8cb206","last_reissued_at":"2026-07-05T09:11:22.269055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:11:22.269055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dongyeop Kang, Jong Inn Park, Minhwa Lee, Ryan Koo, Vipul Raheja, Zae Myung Kim","submitted_at":"2023-09-29T06:53:10Z","abstract_excerpt":"Large Language Models are cognitively biased judges. Large Language Models (LLMs) have recently been shown to be effective as automatic evaluators with simple prompting and in-context learning. In this work, we assemble 15 LLMs of four different size ranges and evaluate their output responses by preference ranking from the other LLMs as evaluators, such as System Star is better than System Square. We then evaluate the quality of ranking outputs introducing the Cognitive Bias Benchmark for LLMs as Evaluators (CoBBLEr), a benchmark to measure six different cognitive biases in LLM evaluation outp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.17012","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.17012/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.17012","created_at":"2026-07-05T09:11:22.269113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.17012v3","created_at":"2026-07-05T09:11:22.269113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.17012","created_at":"2026-07-05T09:11:22.269113+00:00"},{"alias_kind":"pith_short_12","alias_value":"3SMX3ST62SRW","created_at":"2026-07-05T09:11:22.269113+00:00"},{"alias_kind":"pith_short_16","alias_value":"3SMX3ST62SRWMZ6B","created_at":"2026-07-05T09:11:22.269113+00:00"},{"alias_kind":"pith_short_8","alias_value":"3SMX3ST6","created_at":"2026-07-05T09:11:22.269113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":65,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30556","citing_title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27914","citing_title":"Does Capability Transfer to Subjective Behavior -- and Would Our Instruments Tell Us? A Self-Evolving, Trust-by-Construction Evaluation Paradigm","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27921","citing_title":"Show, Don't TELL: Explainable AI-Generated Text Detection","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13076","citing_title":"LLM Evaluators Recognize and Favor Their Own Generations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17554","citing_title":"Evaluating Deep Research Agents on Expert Consulting Work: A Benchmark with Verifiers, Rubrics, and Cognitive Traps","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2404.19737","citing_title":"Better & Faster Large Language Models via Multi-token Prediction","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06996","citing_title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15302","citing_title":"Diagnosing LLM Judge Reliability: Conformal Prediction Sets and Transitivity Violations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02765","citing_title":"U-Define: Designing User Workflows for Hard and Soft Constraints in LLM-Based Planning","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z","json":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z.json","graph_json":"https://pith.science/api/pith-number/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/graph.json","events_json":"https://pith.science/api/pith-number/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/events.json","paper":"https://pith.science/paper/3SMX3ST6"},"agent_actions":{"view_html":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z","download_json":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z.json","view_paper":"https://pith.science/paper/3SMX3ST6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.17012&json=true","fetch_graph":"https://pith.science/api/pith-number/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/graph.json","fetch_events":"https://pith.science/api/pith-number/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/action/storage_attestation","attest_author":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/action/author_attestation","sign_citation":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/action/citation_signature","submit_replication":"https://pith.science/pith/3SMX3ST62SRWMZ6BVSJN6ZAS5Z/action/replication_record"}},"created_at":"2026-07-05T09:11:22.269113+00:00","updated_at":"2026-07-05T09:11:22.269113+00:00"}