{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OLHRQRXA3DPU2PEGQIZGSQEFSV","short_pith_number":"pith:OLHRQRXA","schema_version":"1.0","canonical_sha256":"72cf1846e0d8df4d3c868232694085955d8551775850da11b7fb92f46203c98e","source":{"kind":"arxiv","id":"2403.04814","version":3},"attestation_state":"computed","paper":{"title":"Evaluation of LLMs on Syntax-Aware Code Fill-in-the-Middle Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SE"],"primary_cat":"cs.CL","authors_text":"Alvin Cheung, Linyuan Gong, Mostafa Elhoushi, Sida Wang","submitted_at":"2024-03-07T05:05:56Z","abstract_excerpt":"We introduce Syntax-Aware Fill-In-the-Middle (SAFIM), a new benchmark for evaluating Large Language Models (LLMs) on the code Fill-in-the-Middle (FIM) task. This benchmark focuses on syntax-aware completions of program structures such as code blocks and conditional expressions, and includes 17,720 examples from multiple programming languages, sourced from recent code submissions after April 2022 to minimize data contamination. SAFIM provides a robust framework with various prompt designs and novel syntax-aware post-processing techniques, facilitating accurate and fair comparisons across LLMs. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04814","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-07T05:05:56Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SE"],"title_canon_sha256":"0eb8f39b023e0f9d83766eb82568420c8c8486f936a6d9d93644d8fe5f4f0f83","abstract_canon_sha256":"f8005afdbb3f4d466dcdb38324099a44fa69520c041e8d6b4dd001a0663bf506"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:43.794410Z","signature_b64":"u3eu15jaogE9qzOnDDZOX5odKne2HbV7ykwmdVgRXq5aDGg2JFgftJ/3Xcmj1kZJhIEueLDMGHmlAc16AWTmBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72cf1846e0d8df4d3c868232694085955d8551775850da11b7fb92f46203c98e","last_reissued_at":"2026-07-05T08:35:43.793948Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:43.793948Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of LLMs on Syntax-Aware Code Fill-in-the-Middle Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SE"],"primary_cat":"cs.CL","authors_text":"Alvin Cheung, Linyuan Gong, Mostafa Elhoushi, Sida Wang","submitted_at":"2024-03-07T05:05:56Z","abstract_excerpt":"We introduce Syntax-Aware Fill-In-the-Middle (SAFIM), a new benchmark for evaluating Large Language Models (LLMs) on the code Fill-in-the-Middle (FIM) task. This benchmark focuses on syntax-aware completions of program structures such as code blocks and conditional expressions, and includes 17,720 examples from multiple programming languages, sourced from recent code submissions after April 2022 to minimize data contamination. SAFIM provides a robust framework with various prompt designs and novel syntax-aware post-processing techniques, facilitating accurate and fair comparisons across LLMs. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04814","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04814/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04814","created_at":"2026-07-05T08:35:43.794003+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04814v3","created_at":"2026-07-05T08:35:43.794003+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04814","created_at":"2026-07-05T08:35:43.794003+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLHRQRXA3DPU","created_at":"2026-07-05T08:35:43.794003+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLHRQRXA3DPU2PEG","created_at":"2026-07-05T08:35:43.794003+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLHRQRXA","created_at":"2026-07-05T08:35:43.794003+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04273","citing_title":"Human agency in initial human-AI proof formalization workflows","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25198","citing_title":"Hide to Guide: Learning via Semantic Masking","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03130","citing_title":"Synthetic Hallucinations, Real Gains: Hard Negatives from Frontier Models for FIM Hallucination Mitigation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04894","citing_title":"SynConfRoute: Syntax-Aware Routing for Efficient Code Completion with Small CodeLLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21505","citing_title":"Assessing the Impact of Requirement Ambiguity on LLM-based Function-Level Code Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12186","citing_title":"Qwen2.5-Coder Technical Report","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV","json":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV.json","graph_json":"https://pith.science/api/pith-number/OLHRQRXA3DPU2PEGQIZGSQEFSV/graph.json","events_json":"https://pith.science/api/pith-number/OLHRQRXA3DPU2PEGQIZGSQEFSV/events.json","paper":"https://pith.science/paper/OLHRQRXA"},"agent_actions":{"view_html":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV","download_json":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV.json","view_paper":"https://pith.science/paper/OLHRQRXA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04814&json=true","fetch_graph":"https://pith.science/api/pith-number/OLHRQRXA3DPU2PEGQIZGSQEFSV/graph.json","fetch_events":"https://pith.science/api/pith-number/OLHRQRXA3DPU2PEGQIZGSQEFSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV/action/storage_attestation","attest_author":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV/action/author_attestation","sign_citation":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV/action/citation_signature","submit_replication":"https://pith.science/pith/OLHRQRXA3DPU2PEGQIZGSQEFSV/action/replication_record"}},"created_at":"2026-07-05T08:35:43.794003+00:00","updated_at":"2026-07-05T08:35:43.794003+00:00"}