{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MBXKBWMBUGWPVK5KIFCDV4TTFN","short_pith_number":"pith:MBXKBWMB","schema_version":"1.0","canonical_sha256":"606ea0d981a1acfaabaa41443af2732b7cd7b37b7084fbe9eaf1285a4bad19eb","source":{"kind":"arxiv","id":"2404.01334","version":2},"attestation_state":"computed","paper":{"title":"Augmenting NER Datasets with LLMs: Towards Automated and Refined Annotation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hiroki Naganuma, Kotaro Yoshida, Ryosuke Yamaki, Ryotaro Shimizu, Takafumi Horie, Yoshikazu Ikeda, Yuji Naraki","submitted_at":"2024-03-30T12:13:57Z","abstract_excerpt":"In the field of Natural Language Processing (NLP), Named Entity Recognition (NER) is recognized as a critical technology, employed across a wide array of applications. Traditional methodologies for annotating datasets for NER models are challenged by high costs and variations in dataset quality. This research introduces a novel hybrid annotation approach that synergizes human effort with the capabilities of Large Language Models (LLMs). This approach not only aims to ameliorate the noise inherent in manual annotations, such as omissions, thereby enhancing the performance of NER models, but als"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01334","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-30T12:13:57Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a08552546e51472173d91be5539bbdb58a8739afd91dbdc1b10f36589652f90a","abstract_canon_sha256":"5cb2d499c84d98009eaf0446a693899d56fd038df32144a192d93f19a1c6bd2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:32.928967Z","signature_b64":"2BA2DAPBc9cXRjB6YLMbsDUhhVPwgkn4i7H/vvhEA2jq/5BOllmUJXwcHqwp6ChD/N/Zpemnct2/0cGXFnh3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"606ea0d981a1acfaabaa41443af2732b7cd7b37b7084fbe9eaf1285a4bad19eb","last_reissued_at":"2026-07-05T09:55:32.928481Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:32.928481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Augmenting NER Datasets with LLMs: Towards Automated and Refined Annotation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hiroki Naganuma, Kotaro Yoshida, Ryosuke Yamaki, Ryotaro Shimizu, Takafumi Horie, Yoshikazu Ikeda, Yuji Naraki","submitted_at":"2024-03-30T12:13:57Z","abstract_excerpt":"In the field of Natural Language Processing (NLP), Named Entity Recognition (NER) is recognized as a critical technology, employed across a wide array of applications. Traditional methodologies for annotating datasets for NER models are challenged by high costs and variations in dataset quality. This research introduces a novel hybrid annotation approach that synergizes human effort with the capabilities of Large Language Models (LLMs). This approach not only aims to ameliorate the noise inherent in manual annotations, such as omissions, thereby enhancing the performance of NER models, but als"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01334","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01334","created_at":"2026-07-05T09:55:32.928541+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01334v2","created_at":"2026-07-05T09:55:32.928541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01334","created_at":"2026-07-05T09:55:32.928541+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBXKBWMBUGWP","created_at":"2026-07-05T09:55:32.928541+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBXKBWMBUGWPVK5K","created_at":"2026-07-05T09:55:32.928541+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBXKBWMB","created_at":"2026-07-05T09:55:32.928541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24734","citing_title":"Task Decomposition for Efficient Annotation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN","json":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN.json","graph_json":"https://pith.science/api/pith-number/MBXKBWMBUGWPVK5KIFCDV4TTFN/graph.json","events_json":"https://pith.science/api/pith-number/MBXKBWMBUGWPVK5KIFCDV4TTFN/events.json","paper":"https://pith.science/paper/MBXKBWMB"},"agent_actions":{"view_html":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN","download_json":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN.json","view_paper":"https://pith.science/paper/MBXKBWMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01334&json=true","fetch_graph":"https://pith.science/api/pith-number/MBXKBWMBUGWPVK5KIFCDV4TTFN/graph.json","fetch_events":"https://pith.science/api/pith-number/MBXKBWMBUGWPVK5KIFCDV4TTFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN/action/storage_attestation","attest_author":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN/action/author_attestation","sign_citation":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN/action/citation_signature","submit_replication":"https://pith.science/pith/MBXKBWMBUGWPVK5KIFCDV4TTFN/action/replication_record"}},"created_at":"2026-07-05T09:55:32.928541+00:00","updated_at":"2026-07-05T09:55:32.928541+00:00"}