{"id":"1a730ff5-3cfd-414d-afdc-d3fb9e6cca3e","arxiv_id":"2504.12086","paper_version":1,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":5.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":4,"one_line_summary":"Under sub-exponential reward delays, Delayed NeuralUCB achieves regret O(d~√T logT + d~^{3/2}D+ log^{3/2}T), with D+ depending on the expected delay.","lead":"Delayed NeuralUCB and Delayed NeuralTS adapt neural bandit algorithms to settings where rewards arrive after random, unknown delays, and the paper proves a regret bound whose delay cost does not grow linearly with the horizon. It matters for recommender systems, clinical trials, and any online decision system where feedback is slow.","discovery_kind":"extension","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-16T12:39:47.360289+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}