{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:LEHEWO2B5GHV5Y4XXFFGRVWXYO","short_pith_number":"pith:LEHEWO2B","schema_version":"1.0","canonical_sha256":"590e4b3b41e98f5ee397b94a68d6d7c398d094099e92db8b15b85a80472d25fb","source":{"kind":"arxiv","id":"2203.06482","version":2},"attestation_state":"computed","paper":{"title":"FiNER: Financial Numeric Entity Recognition for XBRL Tagging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eirini Spyropoulou, Georgios Paliouras, Ilias Chalkidis, Ion Androutsopoulos, Lefteris Loukas, Manos Fergadiotis, Prodromos Malakasiotis","submitted_at":"2022-03-12T16:43:57Z","abstract_excerpt":"Publicly traded companies are required to submit periodic reports with eXtensive Business Reporting Language (XBRL) word-level tags. Manually tagging the reports is tedious and costly. We, therefore, introduce XBRL tagging as a new entity extraction task for the financial domain and release FiNER-139, a dataset of 1.1M sentences with gold XBRL tags. Unlike typical entity extraction datasets, FiNER-139 uses a much larger label set of 139 entity types. Most annotated tokens are numeric, with the correct tag per token depending mostly on context, rather than the token itself. We show that subword"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.06482","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-03-12T16:43:57Z","cross_cats_sorted":[],"title_canon_sha256":"da0c3b83b588c5b0697801c5ef013b0921c255f180d48b1a2f9942affa4a1269","abstract_canon_sha256":"2e0a56582465d825a675625507f58b30620744332cf6408365f36bfc06b4438f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:12.374154Z","signature_b64":"26jzRNM2XyUuKVkmGcrQWYrXzt4OUn9riK1m7WJLUPhYULMpDVnSaHz4NEmLao/0Q9ARonIY5jSlQsTJYS0gCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"590e4b3b41e98f5ee397b94a68d6d7c398d094099e92db8b15b85a80472d25fb","last_reissued_at":"2026-07-05T06:15:12.373657Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:12.373657Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FiNER: Financial Numeric Entity Recognition for XBRL Tagging","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eirini Spyropoulou, Georgios Paliouras, Ilias Chalkidis, Ion Androutsopoulos, Lefteris Loukas, Manos Fergadiotis, Prodromos Malakasiotis","submitted_at":"2022-03-12T16:43:57Z","abstract_excerpt":"Publicly traded companies are required to submit periodic reports with eXtensive Business Reporting Language (XBRL) word-level tags. Manually tagging the reports is tedious and costly. We, therefore, introduce XBRL tagging as a new entity extraction task for the financial domain and release FiNER-139, a dataset of 1.1M sentences with gold XBRL tags. Unlike typical entity extraction datasets, FiNER-139 uses a much larger label set of 139 entity types. Most annotated tokens are numeric, with the correct tag per token depending mostly on context, rather than the token itself. We show that subword"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.06482","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.06482/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.06482","created_at":"2026-07-05T06:15:12.373716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.06482v2","created_at":"2026-07-05T06:15:12.373716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.06482","created_at":"2026-07-05T06:15:12.373716+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEHEWO2B5GHV","created_at":"2026-07-05T06:15:12.373716+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEHEWO2B5GHV5Y4X","created_at":"2026-07-05T06:15:12.373716+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEHEWO2B","created_at":"2026-07-05T06:15:12.373716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20650","citing_title":"FinTagging: Benchmarking LLMs for Extracting and Structuring Financial Information","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08886","citing_title":"FinAuditing: A Financial Taxonomy-Structured Multi-Document Benchmark for Evaluating LLMs","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO","json":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO.json","graph_json":"https://pith.science/api/pith-number/LEHEWO2B5GHV5Y4XXFFGRVWXYO/graph.json","events_json":"https://pith.science/api/pith-number/LEHEWO2B5GHV5Y4XXFFGRVWXYO/events.json","paper":"https://pith.science/paper/LEHEWO2B"},"agent_actions":{"view_html":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO","download_json":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO.json","view_paper":"https://pith.science/paper/LEHEWO2B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.06482&json=true","fetch_graph":"https://pith.science/api/pith-number/LEHEWO2B5GHV5Y4XXFFGRVWXYO/graph.json","fetch_events":"https://pith.science/api/pith-number/LEHEWO2B5GHV5Y4XXFFGRVWXYO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO/action/storage_attestation","attest_author":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO/action/author_attestation","sign_citation":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO/action/citation_signature","submit_replication":"https://pith.science/pith/LEHEWO2B5GHV5Y4XXFFGRVWXYO/action/replication_record"}},"created_at":"2026-07-05T06:15:12.373716+00:00","updated_at":"2026-07-05T06:15:12.373716+00:00"}