{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6QVYNFS7IM6CILVHBVPRD3U4EY","short_pith_number":"pith:6QVYNFS7","schema_version":"1.0","canonical_sha256":"f42b86965f433c242ea70d5f11ee9c2636449db26f701ce612db42bc06d6061a","source":{"kind":"arxiv","id":"2410.09038","version":2},"attestation_state":"computed","paper":{"title":"SimpleStrat: Diversifying Language Model Generation with Stratification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Joseph E. Gonzalez, Justin Wong, Michael Luo, Sanjit A. Seshia, Yury Orlovskiy","submitted_at":"2024-10-11T17:54:14Z","abstract_excerpt":"Generating diverse responses from large language models (LLMs) is crucial for applications such as planning/search and synthetic data generation, where diversity provides distinct answers across generations. Prior approaches rely on increasing temperature to increase diversity. However, contrary to popular belief, we show not only does this approach produce lower quality individual generations as temperature increases, but it depends on model's next-token probabilities being similar to the true distribution of answers. We propose SimpleStrat, an alternative approach that uses the language mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09038","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-11T17:54:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f4184232578320b3d672a8008c74db9ba323743b6140f4a94b574031c03b4b71","abstract_canon_sha256":"03286251f7f68600f62abfe049c3e1bc44a646066a8ca2cd2cfe36f607892775"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:05.140041Z","signature_b64":"PZi8RG0EmAfTl2h4HpuBYzS55npuZ+uo19zlqdt7xvk+KtdTKR8gMHtu+PSrtrlpWsmoCCk4ReSy84Pr3/3/BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f42b86965f433c242ea70d5f11ee9c2636449db26f701ce612db42bc06d6061a","last_reissued_at":"2026-07-05T09:20:05.139641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:05.139641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SimpleStrat: Diversifying Language Model Generation with Stratification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Joseph E. Gonzalez, Justin Wong, Michael Luo, Sanjit A. Seshia, Yury Orlovskiy","submitted_at":"2024-10-11T17:54:14Z","abstract_excerpt":"Generating diverse responses from large language models (LLMs) is crucial for applications such as planning/search and synthetic data generation, where diversity provides distinct answers across generations. Prior approaches rely on increasing temperature to increase diversity. However, contrary to popular belief, we show not only does this approach produce lower quality individual generations as temperature increases, but it depends on model's next-token probabilities being similar to the true distribution of answers. We propose SimpleStrat, an alternative approach that uses the language mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09038","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09038","created_at":"2026-07-05T09:20:05.139698+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09038v2","created_at":"2026-07-05T09:20:05.139698+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09038","created_at":"2026-07-05T09:20:05.139698+00:00"},{"alias_kind":"pith_short_12","alias_value":"6QVYNFS7IM6C","created_at":"2026-07-05T09:20:05.139698+00:00"},{"alias_kind":"pith_short_16","alias_value":"6QVYNFS7IM6CILVH","created_at":"2026-07-05T09:20:05.139698+00:00"},{"alias_kind":"pith_short_8","alias_value":"6QVYNFS7","created_at":"2026-07-05T09:20:05.139698+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01697","citing_title":"BARE: Leveraging Base Language Models for Few-Shot Synthetic Data Generation","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY","json":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY.json","graph_json":"https://pith.science/api/pith-number/6QVYNFS7IM6CILVHBVPRD3U4EY/graph.json","events_json":"https://pith.science/api/pith-number/6QVYNFS7IM6CILVHBVPRD3U4EY/events.json","paper":"https://pith.science/paper/6QVYNFS7"},"agent_actions":{"view_html":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY","download_json":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY.json","view_paper":"https://pith.science/paper/6QVYNFS7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09038&json=true","fetch_graph":"https://pith.science/api/pith-number/6QVYNFS7IM6CILVHBVPRD3U4EY/graph.json","fetch_events":"https://pith.science/api/pith-number/6QVYNFS7IM6CILVHBVPRD3U4EY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY/action/storage_attestation","attest_author":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY/action/author_attestation","sign_citation":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY/action/citation_signature","submit_replication":"https://pith.science/pith/6QVYNFS7IM6CILVHBVPRD3U4EY/action/replication_record"}},"created_at":"2026-07-05T09:20:05.139698+00:00","updated_at":"2026-07-05T09:20:05.139698+00:00"}