{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2004:OV4EW4RYIPJAUI5XFQWQ7BLNPN","short_pith_number":"pith:OV4EW4RY","schema_version":"1.0","canonical_sha256":"75784b723843d20a23b72c2d0f856d7b5a3354d89427537b9e5081b9890faed5","source":{"kind":"arxiv","id":"cs/0408007","version":1},"attestation_state":"computed","paper":{"title":"Online convex optimization in the bandit setting: gradient descent without a gradient","license":"","headline":"","cross_cats":["cs.CC"],"primary_cat":"cs.LG","authors_text":"Abraham D. Flaxman, Adam Tauman Kalai, H. Brendan McMahan","submitted_at":"2004-08-02T21:24:41Z","abstract_excerpt":"We consider a the general online convex optimization framework introduced by Zinkevich. In this setting, there is a sequence of convex functions. Each period, we must choose a signle point (from some feasible set) and pay a cost equal to the value of the next function on our chosen point. Zinkevich shows that, if the each function is revealed after the choice is made, then one can achieve vanishingly small regret relative the best single decision chosen in hindsight.\n  We extend this to the bandit setting where we do not find out the entire functions but rather just their value at our chosen p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"cs/0408007","kind":"arxiv","version":1},"metadata":{"license":"","primary_cat":"cs.LG","submitted_at":"2004-08-02T21:24:41Z","cross_cats_sorted":["cs.CC"],"title_canon_sha256":"4b714d2cc5b6ca3577ff1ed58ff9902e1ac2372bd787872ed604dc8d2865a134","abstract_canon_sha256":"5c05d6d7e6d5d0e0646af1fdb42cf06cb8bee7ad734466639d923d5de1427ec2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T14:24:45.601291Z","signature_b64":"pqafj33yb2cdiKRqq0geJctj2+s706fHI3n7OldSbwkt5OO1ZLjmoIakmSw7OUosn7y5Ij0ZGymI0sf2jWmLBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75784b723843d20a23b72c2d0f856d7b5a3354d89427537b9e5081b9890faed5","last_reissued_at":"2026-07-04T14:24:45.600880Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T14:24:45.600880Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Online convex optimization in the bandit setting: gradient descent without a gradient","license":"","headline":"","cross_cats":["cs.CC"],"primary_cat":"cs.LG","authors_text":"Abraham D. Flaxman, Adam Tauman Kalai, H. Brendan McMahan","submitted_at":"2004-08-02T21:24:41Z","abstract_excerpt":"We consider a the general online convex optimization framework introduced by Zinkevich. In this setting, there is a sequence of convex functions. Each period, we must choose a signle point (from some feasible set) and pay a cost equal to the value of the next function on our chosen point. Zinkevich shows that, if the each function is revealed after the choice is made, then one can achieve vanishingly small regret relative the best single decision chosen in hindsight.\n  We extend this to the bandit setting where we do not find out the entire functions but rather just their value at our chosen p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"cs/0408007","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/cs/0408007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"cs/0408007","created_at":"2026-07-04T14:24:45.600938+00:00"},{"alias_kind":"arxiv_version","alias_value":"cs/0408007v1","created_at":"2026-07-04T14:24:45.600938+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.cs/0408007","created_at":"2026-07-04T14:24:45.600938+00:00"},{"alias_kind":"pith_short_12","alias_value":"OV4EW4RYIPJA","created_at":"2026-07-04T14:24:45.600938+00:00"},{"alias_kind":"pith_short_16","alias_value":"OV4EW4RYIPJAUI5X","created_at":"2026-07-04T14:24:45.600938+00:00"},{"alias_kind":"pith_short_8","alias_value":"OV4EW4RY","created_at":"2026-07-04T14:24:45.600938+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":19,"sample":[{"citing_arxiv_id":"2606.23414","citing_title":"Leveraging Similarities in Multi-Armed Bandits","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.14970","citing_title":"Zero-order Parameter-free Optimization for LMO-based Methods: Novel Approach for Efficient Fine-tuning","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08028","citing_title":"Noise-Adaptive High-Probability Regret Bounds for Online Convex Optimization","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06418","citing_title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02869","citing_title":"ZOAF: Towards Efficient Zeroth-Order Optimization for Analog/RF Circuit Design","ref_index":46,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23351","citing_title":"Prudent-Banker: No Extra Fees for Baseline Safety in Adversarial Bandits With and Without Delays","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22191","citing_title":"Bandit Convex Optimization with Gradient Prediction Adaptivity","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2511.01126","citing_title":"Stochastic Regret Guarantees for Online Zeroth- and First-Order Bilevel Optimization","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20515","citing_title":"Online Conformal Prediction with Corrupted Feedback","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15622","citing_title":"Position: Zeroth-Order Optimization in Deep Learning Is Underexplored, Not Underpowered","ref_index":111,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":70,"is_internal_anchor":true},{"citing_arxiv_id":"2501.09732","citing_title":"Inference-Time Scaling for Diffusion Models beyond Scaling Denoising Steps","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17394","citing_title":"Stochastic Zeroth-Order Optimization Under Heavy-Tailed Noise","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2510.26890","citing_title":"Baryon-antibaryon photoproduction cross sections off the proton","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"2511.11308","citing_title":"Policy Optimization for Unknown Systems using Differentiable Model Predictive Control","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2603.06977","citing_title":"NePPO: Near-Potential Policy Optimization for General-Sum Multi-Agent Reinforcement Learning","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2605.09209","citing_title":"Select-then-differentiate: Solving Bilevel Optimization with Manifold Lower-level Solution Sets","ref_index":195,"is_internal_anchor":true},{"citing_arxiv_id":"2605.07565","citing_title":"Ensemble Distributionally Robust Bayesian Optimisation with Continuous Context","ref_index":137,"is_internal_anchor":true},{"citing_arxiv_id":"2605.03065","citing_title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN","json":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN.json","graph_json":"https://pith.science/api/pith-number/OV4EW4RYIPJAUI5XFQWQ7BLNPN/graph.json","events_json":"https://pith.science/api/pith-number/OV4EW4RYIPJAUI5XFQWQ7BLNPN/events.json","paper":"https://pith.science/paper/OV4EW4RY"},"agent_actions":{"view_html":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN","download_json":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN.json","view_paper":"https://pith.science/paper/OV4EW4RY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=cs/0408007&json=true","fetch_graph":"https://pith.science/api/pith-number/OV4EW4RYIPJAUI5XFQWQ7BLNPN/graph.json","fetch_events":"https://pith.science/api/pith-number/OV4EW4RYIPJAUI5XFQWQ7BLNPN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN/action/storage_attestation","attest_author":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN/action/author_attestation","sign_citation":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN/action/citation_signature","submit_replication":"https://pith.science/pith/OV4EW4RYIPJAUI5XFQWQ7BLNPN/action/replication_record"}},"created_at":"2026-07-04T14:24:45.600938+00:00","updated_at":"2026-07-04T14:24:45.600938+00:00"}