{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GOOEFXFPEBEPWVTYQ7OUDCDHTZ","short_pith_number":"pith:GOOEFXFP","schema_version":"1.0","canonical_sha256":"339c42dcaf2048fb567887dd4188679e721a7e786e584d42ab6144f3467e3c44","source":{"kind":"arxiv","id":"2505.10736","version":3},"attestation_state":"computed","paper":{"title":"Model Performance-Guided Evaluation Data Selection for Effective Prompt Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed E. Hassan, Dayi Lin, Shaowei Wang, Ximing Dong","submitted_at":"2025-05-15T22:41:30Z","abstract_excerpt":"Optimizing Large Language Model (LLM) performance requires well-crafted prompts, but manual prompt engineering is labor-intensive and often ineffective. Automated prompt optimization techniques address this challenge but the majority of them rely on randomly selected evaluation subsets, which fail to represent the full dataset, leading to unreliable evaluations and suboptimal prompts. Existing coreset selection methods, designed for LLM benchmarking, are unsuitable for prompt optimization due to challenges in clustering similar samples, high data collection costs, and the unavailability of per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10736","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-15T22:41:30Z","cross_cats_sorted":[],"title_canon_sha256":"d4173b37f62a710ae04b89d7b397484e08dc0e77e6148dbf92156239dda77272","abstract_canon_sha256":"a3f168d2778aafb2fa3178f643735dd3f2bfebbbfc46acfa97975347a813d585"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:39.974478Z","signature_b64":"gk8olNZbE9Xz8eW/FrdgoNPrToYrfYV38RyY9tog7LdALn5U3aXd7iJzTiYt/bsxR8s3bSZSGLylK7edHgEFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"339c42dcaf2048fb567887dd4188679e721a7e786e584d42ab6144f3467e3c44","last_reissued_at":"2026-07-05T11:55:39.974039Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:39.974039Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Performance-Guided Evaluation Data Selection for Effective Prompt Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed E. Hassan, Dayi Lin, Shaowei Wang, Ximing Dong","submitted_at":"2025-05-15T22:41:30Z","abstract_excerpt":"Optimizing Large Language Model (LLM) performance requires well-crafted prompts, but manual prompt engineering is labor-intensive and often ineffective. Automated prompt optimization techniques address this challenge but the majority of them rely on randomly selected evaluation subsets, which fail to represent the full dataset, leading to unreliable evaluations and suboptimal prompts. Existing coreset selection methods, designed for LLM benchmarking, are unsuitable for prompt optimization due to challenges in clustering similar samples, high data collection costs, and the unavailability of per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10736","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10736","created_at":"2026-07-05T11:55:39.974094+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10736v3","created_at":"2026-07-05T11:55:39.974094+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10736","created_at":"2026-07-05T11:55:39.974094+00:00"},{"alias_kind":"pith_short_12","alias_value":"GOOEFXFPEBEP","created_at":"2026-07-05T11:55:39.974094+00:00"},{"alias_kind":"pith_short_16","alias_value":"GOOEFXFPEBEPWVTY","created_at":"2026-07-05T11:55:39.974094+00:00"},{"alias_kind":"pith_short_8","alias_value":"GOOEFXFP","created_at":"2026-07-05T11:55:39.974094+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ","json":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ.json","graph_json":"https://pith.science/api/pith-number/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/graph.json","events_json":"https://pith.science/api/pith-number/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/events.json","paper":"https://pith.science/paper/GOOEFXFP"},"agent_actions":{"view_html":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ","download_json":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ.json","view_paper":"https://pith.science/paper/GOOEFXFP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10736&json=true","fetch_graph":"https://pith.science/api/pith-number/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/action/storage_attestation","attest_author":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/action/author_attestation","sign_citation":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/action/citation_signature","submit_replication":"https://pith.science/pith/GOOEFXFPEBEPWVTYQ7OUDCDHTZ/action/replication_record"}},"created_at":"2026-07-05T11:55:39.974094+00:00","updated_at":"2026-07-05T11:55:39.974094+00:00"}