{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OWX26JF252HLWIHHODYVP76E2G","short_pith_number":"pith:OWX26JF2","schema_version":"1.0","canonical_sha256":"75afaf24baee8ebb20e770f157ffc4d1886d7d0e48fb2a168d34368c16964d2d","source":{"kind":"arxiv","id":"2306.05087","version":2},"attestation_state":"computed","paper":{"title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaoya Jiang, Cunxiang Wang, Hao Chen, Jindong Wang, Linyi Yang, Rui Xie, Shikun Zhang, Wei Ye, Xing Xie, Yidong Wang, Yue Zhang, Zhengran Zeng, Zhuohao Yu","submitted_at":"2023-06-08T10:41:56Z","abstract_excerpt":"Instruction tuning large language models (LLMs) remains a challenging task, owing to the complexity of hyperparameter selection and the difficulty involved in evaluating the tuned models. To determine the optimal hyperparameters, an automatic, robust, and reliable evaluation benchmark is essential. However, establishing such a benchmark is not a trivial task due to the challenges associated with evaluation accuracy and privacy protection. In response to these challenges, we introduce a judge large language model, named PandaLM, which is trained to distinguish the superior model given several L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.05087","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-08T10:41:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"034adee01f4bb1278e0c0aac6b7831eeb0d87281c9edbee5d86614df97783940","abstract_canon_sha256":"28d43162a32de8211594887c76ed16c7e224f04c0e821a9f78f9484f92330a41"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:31.847767Z","signature_b64":"lLjHtfPReCMKekedJKdxaP6YXB/yejVeX71i6aXTJ5LSGS28M3nJ5a45LiPpuOrApaj2Fxpv/Iny+MrdunGVDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75afaf24baee8ebb20e770f157ffc4d1886d7d0e48fb2a168d34368c16964d2d","last_reissued_at":"2026-07-05T08:22:31.847292Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:31.847292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaoya Jiang, Cunxiang Wang, Hao Chen, Jindong Wang, Linyi Yang, Rui Xie, Shikun Zhang, Wei Ye, Xing Xie, Yidong Wang, Yue Zhang, Zhengran Zeng, Zhuohao Yu","submitted_at":"2023-06-08T10:41:56Z","abstract_excerpt":"Instruction tuning large language models (LLMs) remains a challenging task, owing to the complexity of hyperparameter selection and the difficulty involved in evaluating the tuned models. To determine the optimal hyperparameters, an automatic, robust, and reliable evaluation benchmark is essential. However, establishing such a benchmark is not a trivial task due to the challenges associated with evaluation accuracy and privacy protection. In response to these challenges, we introduce a judge large language model, named PandaLM, which is trained to distinguish the superior model given several L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.05087","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.05087/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.05087","created_at":"2026-07-05T08:22:31.847366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.05087v2","created_at":"2026-07-05T08:22:31.847366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.05087","created_at":"2026-07-05T08:22:31.847366+00:00"},{"alias_kind":"pith_short_12","alias_value":"OWX26JF252HL","created_at":"2026-07-05T08:22:31.847366+00:00"},{"alias_kind":"pith_short_16","alias_value":"OWX26JF252HLWIHH","created_at":"2026-07-05T08:22:31.847366+00:00"},{"alias_kind":"pith_short_8","alias_value":"OWX26JF2","created_at":"2026-07-05T08:22:31.847366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08535","citing_title":"When the Judge Changes, So Does the Measurement: Auditing LLM-as-Judge Reliability","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19714","citing_title":"AURA: Adaptive Uncertainty-aware Refinement for LLM-as-a-Judge Auditing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23884","citing_title":"One Year Later...The Harms Persist, But So Do We!","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23884","citing_title":"One Year Later...The Harms Persist, But So Do We!","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12422","citing_title":"Creating and Evaluating K-12 GenAI Assessment Graders Through Context Engineering","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30556","citing_title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00093","citing_title":"Agreement Metrics for LLM-as-Judge Evaluation: What to Report and Why","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03920","citing_title":"Enhancing Instructional Quality: Leveraging Computer-Assisted Textual Analysis to Generate In-Depth Insights from Educational Artifacts","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2405.19088","citing_title":"Cracking the Code of Juxtaposition: Can AI Models Understand the Humorous Contradictions","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23137","citing_title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2305.17926","citing_title":"Large Language Models are not Fair Evaluators","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":245,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07650","citing_title":"A Statistical Framework for Auditing Behavioral Dependence and Induced Bias in LLM Judges","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03179","citing_title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G","json":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G.json","graph_json":"https://pith.science/api/pith-number/OWX26JF252HLWIHHODYVP76E2G/graph.json","events_json":"https://pith.science/api/pith-number/OWX26JF252HLWIHHODYVP76E2G/events.json","paper":"https://pith.science/paper/OWX26JF2"},"agent_actions":{"view_html":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G","download_json":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G.json","view_paper":"https://pith.science/paper/OWX26JF2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.05087&json=true","fetch_graph":"https://pith.science/api/pith-number/OWX26JF252HLWIHHODYVP76E2G/graph.json","fetch_events":"https://pith.science/api/pith-number/OWX26JF252HLWIHHODYVP76E2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G/action/storage_attestation","attest_author":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G/action/author_attestation","sign_citation":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G/action/citation_signature","submit_replication":"https://pith.science/pith/OWX26JF252HLWIHHODYVP76E2G/action/replication_record"}},"created_at":"2026-07-05T08:22:31.847366+00:00","updated_at":"2026-07-05T08:22:31.847366+00:00"}