{"as_of":"2026-08-10T08:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2558e6d9d6d6154b63720d3714242de8055e00797e0a5f88a7f4d6befd570779","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T03:19:01.477082Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.28576/citation-record","integrity":"/paper/2607.28576/integrity","json":"/paper/2607.28576/citation-record.json","paper":"/paper/2607.28576"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.09687","last_updated":"2024-02-06T18:00:18Z","snapshot_observed_at":"2026-08-06T03:16:16.175894Z","submitted_at":"2023-08-18T17:29:23Z","title":"Graph of Thoughts: Solving Elaborate Problems with Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09687","snapshot_observed_at":"2026-07-31T03:18:59.572680Z","title":"Graph of thoughts: Solving elaborate problems with large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.572680Z"},"links":{"cited_paper":"/paper/2308.09687","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:ae017d6cb16f905d2adb0abd33a02bcf8ed640a756234e8b71441aec8aefdf55","observation_id":"ce7f7f41-d48b-4e80-8c9d-4f2eacf33c24","resolution":{"observed_at":"2026-07-31T03:18:59.572680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-07-31T03:18:59.616128Z","title":"Le, Christopher R´ e, and Azalia Mirhoseini","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.616128Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:f92c69aae9a7314b9ff9f81687f5789c4e452b49a58c86c4f363f67ce5259fec","observation_id":"e21a5e66-a140-435b-8796-cf71034a8cf9","resolution":{"observed_at":"2026-07-31T03:18:59.616128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:18:59.652283Z","title":"Debate or vote: Which yields better decisions in multi-agent large lan- guage models? InAdvances in Neural Information Processing Systems (NeurIPS), Spotlight,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.652283Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:da9d5dc18fabdfb5d6bf1e508dc49250e08c1cb88fb0e125efb31d2180429ad6","observation_id":"f6c28381-4b71-42ca-9d6d-8a16ce472e79","resolution":{"observed_at":"2026-07-31T03:18:59.652283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-31T03:18:59.690707Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.690707Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:248efd0276f6c268381662dcb8ec82ee72cf3d70c352daca43b9451fdcfb214e","observation_id":"99d3e645-2e73-4760-a8ba-f84dba51c93b","resolution":{"observed_at":"2026-07-31T03:18:59.690707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14325","last_updated":"2023-05-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T17:55:11Z","title":"Improving Factuality and Reasoning in Language Models through Multiagent Debate","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14325","snapshot_observed_at":"2026-07-31T03:18:59.728716Z","title":"Tenenbaum, and Igor Mordatch","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.728716Z"},"links":{"cited_paper":"/paper/2305.14325","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:615d80eb3f857ec981b4f039a32edd5c49659a949ae7f9c94425227da02b5f54","observation_id":"1cacb16a-b6cb-4228-93e2-c3c31f6b526e","resolution":{"observed_at":"2026-07-31T03:18:59.728716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:18:59.769984Z","title":"Bootstrap methods: Another look at the jackknife.The Annals of Statistics, 7(1):1–26, 1979","venue":null,"work_id":null,"year":1979},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.769984Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:0def1ea1cacdc49e371c8fd3317138cc2e753fdf623b0508b50f7c470645faed","observation_id":"86cb38bb-1d49-4a79-af60-a372adbf204d","resolution":{"observed_at":"2026-07-31T03:18:59.769984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:18:59.809324Z","title":"llama.cpp, 2026.https://github.com/ ggml-org/llama.cpp","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.809324Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:30745aab2b83af676b84c86c588c02166f9c5e111302649058e2139e3bb90cde","observation_id":"1f741c85-fea5-4a71-a7e4-f59f0155ce51","resolution":{"observed_at":"2026-07-31T03:18:59.809324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-07-31T03:18:59.819809Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.819809Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:5df8efe9e086ba34d423048f602be95eb760b69e882959b1c8a7bb77b2f4edb4","observation_id":"c599eef7-51cb-4aaa-8c3e-763040c9b49d","resolution":{"observed_at":"2026-07-31T03:18:59.819809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:18:59.861811Z","title":"A simple sequentially rejective multiple test procedure.Scandinavian Journal of Statistics, 6(2):65–70, 1979","venue":null,"work_id":null,"year":1979},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.861811Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:3b526e41e89382b9093d39a2f4817ffda3f929004b6b357365383d5487b269ab","observation_id":"34e9b2d7-162d-4ef6-b3ca-5ae0840f0b3d","resolution":{"observed_at":"2026-07-31T03:18:59.861811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06457","last_updated":"2024-08-14T02:41:48Z","snapshot_observed_at":"2026-07-06T17:28:02.037844Z","submitted_at":"2024-02-09T15:02:56Z","title":"V-STaR: Training Verifiers for Self-Taught Reasoners","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06457","snapshot_observed_at":"2026-07-31T03:18:59.902163Z","title":"V-star: Training verifiers for self-taught reasoners.arXiv preprint arXiv:2402.06457, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.902163Z"},"links":{"cited_paper":"/paper/2402.06457","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:b7fcceeb3c0f4659110412929d02a54cf57e468d90ca54fb4e4d1b6109cfb255","observation_id":"cdb006de-9c05-46a9-85ee-f7926df310b8","resolution":{"observed_at":"2026-07-31T03:18:59.902163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01798","last_updated":"2024-03-14T04:27:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T04:56:12Z","title":"Large Language Models Cannot Self-Correct Reasoning Yet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01798","snapshot_observed_at":"2026-07-31T03:18:59.943645Z","title":"Large language models cannot self-correct reasoning yet","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.943645Z"},"links":{"cited_paper":"/paper/2310.01798","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:d7684ce5110426c7e2b3ea45ccfa3c13e21311ee51eb2140f78dbde7aa9dd410","observation_id":"4f3ed3f3-df60-41e2-9d5d-44f54f1414e6","resolution":{"observed_at":"2026-07-31T03:18:59.943645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01297","last_updated":"2024-12-03T19:14:06Z","snapshot_observed_at":"2026-08-02T06:30:19.601023Z","submitted_at":"2024-06-03T13:05:46Z","title":"When Can LLMs Actually Correct Their Own Mistakes? A Critical Survey of Self-Correction of LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01297","snapshot_observed_at":"2026-07-31T03:18:59.984759Z","title":"When can llms actu- ally correct their own mistakes? a critical survey of self-correction of llms.arXiv preprint arXiv:2406.01297, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-31T03:18:59.984759Z"},"links":{"cited_paper":"/paper/2406.01297","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:9506e11613c53a05a096f3034c5708513824258713cf132026613784749c41ac","observation_id":"7a3fa1e7-b0e4-43ca-862e-76b7460d077c","resolution":{"observed_at":"2026-07-31T03:18:59.984759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:19:00.058675Z","title":"Scalable best-of-n selection for large language models via self-certainty.arXiv preprint arXiv:2502.18581, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.058675Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:3d4b4e7bc9fdb04b6d38d14f6b64b104b39524b98df007dc377c909c8feab043","observation_id":"d2fa40f4-9416-4a92-9d85-da7ad8d1a133","resolution":{"observed_at":"2026-07-31T03:19:00.058675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.11916","last_updated":"2023-01-29T05:14:17Z","snapshot_observed_at":"2026-08-07T04:02:08.445660Z","submitted_at":"2022-05-24T09:22:26Z","title":"Large Language Models are Zero-Shot Reasoners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.11916","snapshot_observed_at":"2026-07-31T03:19:00.116288Z","title":"Large language models are zero-shot reasoners","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.116288Z"},"links":{"cited_paper":"/paper/2205.11916","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:26984afd4e119c03fd19b591752a9355ec2813f20f472d54ee2590148131fc30","observation_id":"3158fbd8-1a35-4f98-8b9b-1f9b132e1cef","resolution":{"observed_at":"2026-07-31T03:19:00.116288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-07-31T03:19:00.187681Z","title":"Let’s verify step by step","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.187681Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:6c4dcf8b35924c59c8edc852c3210542c903a235e5df75be98cd3d5ea873734b","observation_id":"6d307dc6-a7a9-4bd8-a21d-fdc458ca8737","resolution":{"observed_at":"2026-07-31T03:19:00.187681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17651","last_updated":"2023-05-25T19:13:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T18:30:01Z","title":"Self-Refine: Iterative Refinement with Self-Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17651","snapshot_observed_at":"2026-07-31T03:19:00.237462Z","title":"Self-refine: Iterative refinement with self-feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.237462Z"},"links":{"cited_paper":"/paper/2303.17651","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:2070d09b4d3ac8a4e877bd1c4dbf58d0957aefd5f2fb3d33c5958b8b7d686cd3","observation_id":"73fc699e-70fd-4600-b3f9-f9ee9b52203e","resolution":{"observed_at":"2026-07-31T03:19:00.237462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00640","last_updated":"2024-11-01T14:57:16Z","snapshot_observed_at":"2026-08-07T10:39:44.274910Z","submitted_at":"2024-11-01T14:57:16Z","title":"Adding Error Bars to Evals: A Statistical Approach to Language Model Evaluations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00640","snapshot_observed_at":"2026-07-31T03:19:00.303898Z","title":"Adding error bars to evals: A statistical approach to language model evaluations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.303898Z"},"links":{"cited_paper":"/paper/2411.00640","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:e91b36d053276bb98ed9c28a95ea40823b7b4a33a9114b5fdd670a8eeda04232","observation_id":"b41f381c-8c5c-41d9-8aef-74bdff93efee","resolution":{"observed_at":"2026-07-31T03:19:00.303898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.13378","last_updated":"2026-07-15T02:15:35Z","snapshot_observed_at":"2026-08-06T18:09:17.674081Z","submitted_at":"2026-07-15T02:15:35Z","title":"Fair on the Surface: Transaction-Ordering Bias and MEV in Mysticeti DAG-based BFT Protocol","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.13378","snapshot_observed_at":"2026-07-31T03:19:00.364677Z","title":"Fair on the surface: Transaction-ordering bias and mev in mysticeti dag-based bft protocol.arXiv preprint arXiv:2607.13378, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.364677Z"},"links":{"cited_paper":"/paper/2607.13378","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:ebc2ceba582fe0fd14b7a1fe6e0c6803623ae90bed85aef2a4cb6fb179e67bf9","observation_id":"3f3b467a-fdfb-4b76-a18f-a7ceeddc2b93","resolution":{"observed_at":"2026-07-31T03:19:00.364677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-07-06T20:29:11.710285Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-07-31T03:19:00.429895Z","title":"s1: Simple test-time scaling.arXiv preprint arXiv:2501.19393, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.429895Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:39a8fec8c4c9939e1ef0892ba9efaa98c8b1fb4c3ec9ecae7a9f3d6c0c245979","observation_id":"e40ff60f-e0e0-4c29-9860-597da010867b","resolution":{"observed_at":"2026-07-31T03:19:00.429895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.14109","last_updated":"2026-05-07T21:11:51Z","snapshot_observed_at":"2026-08-08T06:32:48.904917Z","submitted_at":"2026-05-07T21:11:51Z","title":"Simplicity Paradox: Debunking myths about prompting and datasets for LLM evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.14109","snapshot_observed_at":"2026-07-31T03:19:00.468509Z","title":"Simplicity paradox: Debunking myths about prompting and datasets for llm evaluation.arXiv preprint arXiv:2607.14109, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.468509Z"},"links":{"cited_paper":"/paper/2607.14109","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:076b8c2f0d06c5c6a05a2f736a02bd9dc2131035125013c2d3933dc84d6ee792","observation_id":"c09d33ea-403f-4738-85b1-ac4435c38fb5","resolution":{"observed_at":"2026-07-31T03:19:00.468509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-07-31T03:19:00.517313Z","title":"Qwen2.5 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.517313Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:576673a0ddb9b43d93cb1075b2ffc12437b2877aaa9d2c6f516798901e87a48d","observation_id":"849ec5dc-5993-463b-aab9-21ca3305c913","resolution":{"observed_at":"2026-07-31T03:19:00.517313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:19:00.567426Z","title":"The sequential edge: Inverse-entropy voting beats parallel self-consistency at matched compute.arXiv preprint arXiv:2511.02309, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.567426Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:0d0a1bdef75b10b5873198e701b5ef81f87096609564188e1be9b7634103cca7","observation_id":"ff981305-90ed-49c5-8259-b0a78b552d41","resolution":{"observed_at":"2026-07-31T03:19:00.567426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11366","last_updated":"2023-10-10T05:21:45Z","snapshot_observed_at":"2026-07-06T15:05:53.556198Z","submitted_at":"2023-03-20T18:08:50Z","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11366","snapshot_observed_at":"2026-07-31T03:19:00.608216Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.608216Z"},"links":{"cited_paper":"/paper/2303.11366","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:7bf9c32fed4994aa9c868952f98dbe15b40d8042067cc0f276019c1748c0e868","observation_id":"2a636804-bd97-416c-afc8-45d29c695e58","resolution":{"observed_at":"2026-07-31T03:19:00.608216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-07-31T03:19:00.644343Z","title":"Scaling llm test-time compute opti- mally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.644343Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:50bcb5f1ce3ebc0e983f471714c1dc52cb9803b4bd1b19555f5cd2ffce2b33f8","observation_id":"b3d58206-ae11-42d9-85a2-f42d2964c6e5","resolution":{"observed_at":"2026-07-31T03:19:00.644343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:19:00.700425Z","title":"Inference scaling flaws: The limits of llm resampling with imperfect verifiers.arXiv preprint arXiv:2411.17501, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.700425Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:7c1ad4c27204632437d583e87aaa7d2a0cfcd3c194caf22039bcd5e4fe132e63","observation_id":"3e18785f-cc04-4bf4-b8eb-5619454463e1","resolution":{"observed_at":"2026-07-31T03:19:00.700425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.02460","last_updated":"2026-04-11T23:40:49Z","snapshot_observed_at":"2026-07-06T22:51:48.992363Z","submitted_at":"2026-04-02T18:47:48Z","title":"Single-Agent LLMs Outperform Multi-Agent Systems on Multi-Hop Reasoning Under Equal Thinking Token Budgets","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.02460","snapshot_observed_at":"2026-07-31T03:19:00.768803Z","title":"Single-agent llms outperform multi-agent systems on multi-hop reasoning under equal thinking token budgets.arXiv preprint arXiv:2604.02460, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.768803Z"},"links":{"cited_paper":"/paper/2604.02460","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:0c1e795672d3ffa39a2835d206e6e9aced75a53b9d329ec86c6cad20d5135fe6","observation_id":"68fa443e-7841-459f-814a-55147701f9bf","resolution":{"observed_at":"2026-07-31T03:19:00.768803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06461","last_updated":"2024-06-15T01:59:02Z","snapshot_observed_at":"2026-08-09T22:43:43.316157Z","submitted_at":"2024-06-10T16:55:08Z","title":"Reasoning in Token Economies: Budget-Aware Evaluation of LLM Reasoning Strategies","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06461","snapshot_observed_at":"2026-07-31T03:19:00.810695Z","title":"Reasoning in token economies: Budget-aware evaluation of llm reasoning strategies","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.810695Z"},"links":{"cited_paper":"/paper/2406.06461","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:50226d193f3a9cf42c003dc7b30a504d4dc21852e2f1b4d338053bf468fd3edf","observation_id":"9e3773e0-13b6-4392-96de-f30f6bf94230","resolution":{"observed_at":"2026-07-31T03:19:00.810695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:19:00.887948Z","title":"Plan-and-solve prompting: Improving zero-shot chain-of-thought reasoning by large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.887948Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:85b47322630b24d749b661953fe90f3bf028753bd47cbe891d65817b825071d9","observation_id":"1ab401c6-35bf-435e-9d8b-717772884b58","resolution":{"observed_at":"2026-07-31T03:19:00.887948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-07-31T03:19:01.035096Z","title":"Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.035096Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:b1b09cfbb9e54de12409a942fe394a0e4d1b9258559b077fca7db9095e13f640","observation_id":"bb48a105-620c-44d4-8ad4-a06fb29c3e07","resolution":{"observed_at":"2026-07-31T03:19:01.035096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-07-31T03:19:01.111637Z","title":"Chi, Quoc V","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.111637Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:df7cf7e445b98e3a7b837da006d44bc46e0e6a16fe4d05d2ad9f188348a25d6a","observation_id":"8701c473-0ea7-40c1-b40a-557549e12a40","resolution":{"observed_at":"2026-07-31T03:19:01.111637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-08-09T23:53:42.648697Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-07-31T03:19:01.177346Z","title":"Inference scaling laws: An empirical analysis of compute-optimal inference for problem-solving with language models.arXiv preprint arXiv:2408.00724, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.177346Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:a9eaeb66c2d1daee8e2cca6fd7ba4a44f8e6465b56ef348c76911c363ad2e1dc","observation_id":"f502b14f-dbfa-4269-9b97-4981a9bb318b","resolution":{"observed_at":"2026-07-31T03:19:01.177346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10601","last_updated":"2023-12-03T22:50:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T23:16:17Z","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10601","snapshot_observed_at":"2026-07-31T03:19:01.220557Z","title":"Griffiths, Yuan Cao, and Karthik Narasimhan","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.220557Z"},"links":{"cited_paper":"/paper/2305.10601","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:816445d7b0f824cb1e8f6246fba7c8336e8b21885b9696dc4728d235cf6ed8e2","observation_id":"b3d636e9-6db7-40a2-88d0-9c5568733379","resolution":{"observed_at":"2026-07-31T03:19:01.220557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T03:19:01.285736Z","title":"Incentivizing llms to self-verify their answers","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.285736Z"},"links":{"citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:1c360507d51fdabe1b796f6aa7f5850706f4122ff7a1127ec5155b42b25cc91c","observation_id":"19fe31c5-9bf0-46b5-b97d-5da5dc5a9d70","resolution":{"observed_at":"2026-07-31T03:19:01.285736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09797","last_updated":"2024-10-07T04:28:04Z","snapshot_observed_at":"2026-08-06T05:02:00.041176Z","submitted_at":"2023-04-19T16:29:48Z","title":"Progressive-Hint Prompting Improves Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09797","snapshot_observed_at":"2026-07-31T03:19:01.329887Z","title":"Progressive-hint prompt- ing improves reasoning in large language models.arXiv preprint arXiv:2304.09797, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.329887Z"},"links":{"cited_paper":"/paper/2304.09797","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:3966ffd48bbae8e102684f4c7ef5e06e8cc348f7ea655229f65d77fcb8eca440","observation_id":"92ca9c9d-b4e6-48ea-8da8-f6c54236be20","resolution":{"observed_at":"2026-07-31T03:19:01.329887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-07-31T03:19:01.409610Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.409610Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:3063227f238462cf9c39e65a91c075a385576e841592790675514199476bc5e0","observation_id":"558abeb0-e40f-4fde-bca7-0c93e7a6f5f2","resolution":{"observed_at":"2026-07-31T03:19:01.409610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.10625","last_updated":"2023-04-16T22:08:08Z","snapshot_observed_at":"2026-08-06T09:00:42.886249Z","submitted_at":"2022-05-21T15:34:53Z","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.10625","snapshot_observed_at":"2026-07-31T03:19:01.477082Z","title":"Let’s first understand the problem and devise a plan to solve it. Then carry out the plan step by step","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:01.477082Z"},"links":{"cited_paper":"/paper/2205.10625","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:7d3c7298d07b695f5121a8715a8e25dbdb2c6851e96413ebab6a1d5827034b19","observation_id":"9f9128a1-3ea9-48a0-85e1-d1c939428a48","resolution":{"observed_at":"2026-07-31T03:19:01.477082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04091","last_updated":"2023-05-26T07:06:48Z","snapshot_observed_at":"2026-07-06T15:24:07.662207Z","submitted_at":"2023-05-06T16:34:37Z","title":"Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.04091","snapshot_observed_at":"2026-07-31T03:19:00.973897Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-07-31T03:19:00.973897Z"},"links":{"cited_paper":"/paper/2305.04091","citing_paper":"/paper/2607.28576"},"observation_digest":"sha256:a87d2a5c6b33a3af4abf610f644cf6a92bade1826d806e8d35246553903c0e24","observation_id":"b1bea29f-2a0c-4cc4-880a-5c2a2e311368","resolution":{"observed_at":"2026-07-31T03:19:00.973897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.28576","last_updated":"2026-07-30T17:38:23Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T07:46:35.369681Z","submitted_at":"2026-07-30T17:38:23Z","title":"Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":37,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 0 inbound Pith citation observations for arXiv:2607.28576."}