{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:EAIFARRAS7FEXF6J7A66SKKSIH","short_pith_number":"pith:EAIFARRA","schema_version":"1.0","canonical_sha256":"201050462097ca4b97c9f83de9295241c120c8cdd9d1c4c234b9999121d9a252","source":{"kind":"arxiv","id":"1908.04319","version":2},"attestation_state":"computed","paper":{"title":"Neural Text Generation with Unlikelihood Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Emily Dinan, Ilia Kulikov, Jason Weston, Kyunghyun Cho, Sean Welleck, Stephen Roller","submitted_at":"2019-08-12T18:09:04Z","abstract_excerpt":"Neural text generation is a key tool in natural language applications, but it is well known there are major problems at its core. In particular, standard likelihood training and decoding leads to dull and repetitive outputs. While some post-hoc fixes have been proposed, in particular top-$k$ and nucleus sampling, they do not address the fact that the token-level probabilities predicted by the model are poor. In this paper we show that the likelihood objective itself is at fault, resulting in a model that assigns too much probability to sequences containing repeats and frequent words, unlike th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.04319","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-12T18:09:04Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"ca328ae9b9ea70c9060ef8444192bf495bc752e29b0aee28d48e47fff71f8502","abstract_canon_sha256":"5b28c5e665cba7c5a0641931007cb50859013c4b1ae741cb82067726d95a0518"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:07:37.143529Z","signature_b64":"djFY9lq8/N+Nb41aa5hPEkqfUAfHdTP0GNj9nP78V5WdsaMdy5DmJB87W9oGZsrF8/oVW9NhcwHJcT61hlCrAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"201050462097ca4b97c9f83de9295241c120c8cdd9d1c4c234b9999121d9a252","last_reissued_at":"2026-07-05T00:07:37.143055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:07:37.143055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Text Generation with Unlikelihood Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Emily Dinan, Ilia Kulikov, Jason Weston, Kyunghyun Cho, Sean Welleck, Stephen Roller","submitted_at":"2019-08-12T18:09:04Z","abstract_excerpt":"Neural text generation is a key tool in natural language applications, but it is well known there are major problems at its core. In particular, standard likelihood training and decoding leads to dull and repetitive outputs. While some post-hoc fixes have been proposed, in particular top-$k$ and nucleus sampling, they do not address the fact that the token-level probabilities predicted by the model are poor. In this paper we show that the likelihood objective itself is at fault, resulting in a model that assigns too much probability to sequences containing repeats and frequent words, unlike th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.04319","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.04319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.04319","created_at":"2026-07-05T00:07:37.143112+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.04319v2","created_at":"2026-07-05T00:07:37.143112+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.04319","created_at":"2026-07-05T00:07:37.143112+00:00"},{"alias_kind":"pith_short_12","alias_value":"EAIFARRAS7FE","created_at":"2026-07-05T00:07:37.143112+00:00"},{"alias_kind":"pith_short_16","alias_value":"EAIFARRAS7FEXF6J","created_at":"2026-07-05T00:07:37.143112+00:00"},{"alias_kind":"pith_short_8","alias_value":"EAIFARRA","created_at":"2026-07-05T00:07:37.143112+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02052","citing_title":"Mitigating Package Hallucinations in Large Language Models via Model Editing","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09587","citing_title":"Seeing the Hivemind: A Consensus-Aware Interaction Technique for Mitigating AI Homogenization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30963","citing_title":"AMix-2: Establishing Protein as a Native Modality in Large Language Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01237","citing_title":"The Differences Between Direct Alignment Algorithms are a Blur","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16462","citing_title":"Asking Back: Interaction-Layer Antidistillation Watermarks","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07199","citing_title":"Agent Q: Advanced Reasoning and Learning for Autonomous AI Agents","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2009.01325","citing_title":"Learning to summarize from human feedback","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2402.11411","citing_title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"1909.05858","citing_title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07691","citing_title":"ORPO: Monolithic Preference Optimization without Reference Model","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11317","citing_title":"SOMA: Efficient Multi-turn LLM Serving via Small Language Model","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09995","citing_title":"Annotations Mitigate Post-Training Mode Collapse","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10466","citing_title":"Self-Attention as a Covariance Readout: A Unified View of In-Context Learning and Repetition","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07977","citing_title":"Self-Play Enhancement via Advantage-Weighted Refinement in Online Federated LLM Fine-Tuning with Real-Time Feedback","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2305.18290","citing_title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17323","citing_title":"A Universal Avoidance Method for Diverse Multi-branch Generation","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH","json":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH.json","graph_json":"https://pith.science/api/pith-number/EAIFARRAS7FEXF6J7A66SKKSIH/graph.json","events_json":"https://pith.science/api/pith-number/EAIFARRAS7FEXF6J7A66SKKSIH/events.json","paper":"https://pith.science/paper/EAIFARRA"},"agent_actions":{"view_html":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH","download_json":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH.json","view_paper":"https://pith.science/paper/EAIFARRA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.04319&json=true","fetch_graph":"https://pith.science/api/pith-number/EAIFARRAS7FEXF6J7A66SKKSIH/graph.json","fetch_events":"https://pith.science/api/pith-number/EAIFARRAS7FEXF6J7A66SKKSIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH/action/storage_attestation","attest_author":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH/action/author_attestation","sign_citation":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH/action/citation_signature","submit_replication":"https://pith.science/pith/EAIFARRAS7FEXF6J7A66SKKSIH/action/replication_record"}},"created_at":"2026-07-05T00:07:37.143112+00:00","updated_at":"2026-07-05T00:07:37.143112+00:00"}