{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:BOC52NC6GR7EQNWOLTBNDETT2P","short_pith_number":"pith:BOC52NC6","schema_version":"1.0","canonical_sha256":"0b85dd345e347e4836ce5cc2d19273d3e88dea3af53cf20b77f3c5e898826e90","source":{"kind":"arxiv","id":"2211.11890","version":1},"attestation_state":"computed","paper":{"title":"TEMPERA: Test-Time Prompting via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dale Schuurmans, Denny Zhou, Joseph E. Gonzalez, Tianjun Zhang, Xuezhi Wang","submitted_at":"2022-11-21T22:38:20Z","abstract_excerpt":"Careful prompt design is critical to the use of large language models in zero-shot or few-shot learning. As a consequence, there is a growing interest in automated methods to design optimal prompts. In this work, we propose Test-time Prompt Editing using Reinforcement learning (TEMPERA). In contrast to prior prompt generation methods, TEMPERA can efficiently leverage prior knowledge, is adaptive to different queries and provides an interpretable prompt for every query. To achieve this, we design a novel action space that allows flexible editing of the initial prompts covering a wide set of com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.11890","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-21T22:38:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8b01c6c9d447e2edd488bfbf46766fcce1416752721a20a7b5fa8b23f0b5cbf3","abstract_canon_sha256":"d2da465c8e9b7c7cc05544a7c731ac03da5fea6ff8d9f7f3c96be37ab48e091e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:18:10.578028Z","signature_b64":"l53xCr2ENpnej9cwTTOIdb5H9w1PZzjTBW7sP21WMhO7msAwwRqBPUkjxOnD6jjqh4PGVzoA9auovihJdzl8Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b85dd345e347e4836ce5cc2d19273d3e88dea3af53cf20b77f3c5e898826e90","last_reissued_at":"2026-07-05T05:18:10.577475Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:18:10.577475Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TEMPERA: Test-Time Prompting via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dale Schuurmans, Denny Zhou, Joseph E. Gonzalez, Tianjun Zhang, Xuezhi Wang","submitted_at":"2022-11-21T22:38:20Z","abstract_excerpt":"Careful prompt design is critical to the use of large language models in zero-shot or few-shot learning. As a consequence, there is a growing interest in automated methods to design optimal prompts. In this work, we propose Test-time Prompt Editing using Reinforcement learning (TEMPERA). In contrast to prior prompt generation methods, TEMPERA can efficiently leverage prior knowledge, is adaptive to different queries and provides an interpretable prompt for every query. To achieve this, we design a novel action space that allows flexible editing of the initial prompts covering a wide set of com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.11890","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.11890/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.11890","created_at":"2026-07-05T05:18:10.577534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.11890v1","created_at":"2026-07-05T05:18:10.577534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.11890","created_at":"2026-07-05T05:18:10.577534+00:00"},{"alias_kind":"pith_short_12","alias_value":"BOC52NC6GR7E","created_at":"2026-07-05T05:18:10.577534+00:00"},{"alias_kind":"pith_short_16","alias_value":"BOC52NC6GR7EQNWO","created_at":"2026-07-05T05:18:10.577534+00:00"},{"alias_kind":"pith_short_8","alias_value":"BOC52NC6","created_at":"2026-07-05T05:18:10.577534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04661","citing_title":"CRAFT: Cost-aware Refinement And Front-aware Tuning of Prompts","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31958","citing_title":"Adapting Generalist Robot Policies with Semantic Reinforcement Learning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26655","citing_title":"Why Prompt Optimization Works, and Why It Sometimes Doesn't: A Causal-Inspired Edit-Level Analysis","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10477","citing_title":"PEEM: Prompt Engineering Evaluation Metrics for Interpretable Joint Evaluation of Prompts and Responses","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P","json":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P.json","graph_json":"https://pith.science/api/pith-number/BOC52NC6GR7EQNWOLTBNDETT2P/graph.json","events_json":"https://pith.science/api/pith-number/BOC52NC6GR7EQNWOLTBNDETT2P/events.json","paper":"https://pith.science/paper/BOC52NC6"},"agent_actions":{"view_html":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P","download_json":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P.json","view_paper":"https://pith.science/paper/BOC52NC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.11890&json=true","fetch_graph":"https://pith.science/api/pith-number/BOC52NC6GR7EQNWOLTBNDETT2P/graph.json","fetch_events":"https://pith.science/api/pith-number/BOC52NC6GR7EQNWOLTBNDETT2P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P/action/storage_attestation","attest_author":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P/action/author_attestation","sign_citation":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P/action/citation_signature","submit_replication":"https://pith.science/pith/BOC52NC6GR7EQNWOLTBNDETT2P/action/replication_record"}},"created_at":"2026-07-05T05:18:10.577534+00:00","updated_at":"2026-07-05T05:18:10.577534+00:00"}