{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ICVBGSN7ERMG6JCWT4X3K52KFH","short_pith_number":"pith:ICVBGSN7","schema_version":"1.0","canonical_sha256":"40aa1349bf24586f24569f2fb5774a29c1a483d4e2ca28c28c84021c680615a0","source":{"kind":"arxiv","id":"2412.05137","version":1},"attestation_state":"computed","paper":{"title":"Can Large Language Models Serve as Effective Classifiers for Hierarchical Multi-Label Classification of Scientific Documents at Industrial Scale?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arian Askari, Michael Parsons, Sarah Fancher, Seyed Amin Tabatabaei","submitted_at":"2024-12-06T15:51:22Z","abstract_excerpt":"We address the task of hierarchical multi-label classification (HMC) of scientific documents at an industrial scale, where hundreds of thousands of documents must be classified across thousands of dynamic labels. The rapid growth of scientific publications necessitates scalable and efficient methods for classification, further complicated by the evolving nature of taxonomies--where new categories are introduced, existing ones are merged, and outdated ones are deprecated. Traditional machine learning approaches, which require costly retraining with each taxonomy update, become impractical due t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.05137","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-12-06T15:51:22Z","cross_cats_sorted":[],"title_canon_sha256":"9a17e1be5015cbf7113ead56a29d989f3d4fbd57ffc46f8d79302d505c2bdf4c","abstract_canon_sha256":"26f9046e13344bda19047860391071d756c07306f04efe5fe5b0d3f708736aa7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:28.956236Z","signature_b64":"vkm2u5s74jikdnusPJPQOpdzAy21gb3jwZyWmucVi54EtCLd14q6kZyY6qT6FteMhpvnSckIhcxiId4LQhZ3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40aa1349bf24586f24569f2fb5774a29c1a483d4e2ca28c28c84021c680615a0","last_reissued_at":"2026-07-05T09:45:28.955731Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:28.955731Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Serve as Effective Classifiers for Hierarchical Multi-Label Classification of Scientific Documents at Industrial Scale?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arian Askari, Michael Parsons, Sarah Fancher, Seyed Amin Tabatabaei","submitted_at":"2024-12-06T15:51:22Z","abstract_excerpt":"We address the task of hierarchical multi-label classification (HMC) of scientific documents at an industrial scale, where hundreds of thousands of documents must be classified across thousands of dynamic labels. The rapid growth of scientific publications necessitates scalable and efficient methods for classification, further complicated by the evolving nature of taxonomies--where new categories are introduced, existing ones are merged, and outdated ones are deprecated. Traditional machine learning approaches, which require costly retraining with each taxonomy update, become impractical due t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.05137","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.05137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.05137","created_at":"2026-07-05T09:45:28.955797+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.05137v1","created_at":"2026-07-05T09:45:28.955797+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.05137","created_at":"2026-07-05T09:45:28.955797+00:00"},{"alias_kind":"pith_short_12","alias_value":"ICVBGSN7ERMG","created_at":"2026-07-05T09:45:28.955797+00:00"},{"alias_kind":"pith_short_16","alias_value":"ICVBGSN7ERMG6JCW","created_at":"2026-07-05T09:45:28.955797+00:00"},{"alias_kind":"pith_short_8","alias_value":"ICVBGSN7","created_at":"2026-07-05T09:45:28.955797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH","json":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH.json","graph_json":"https://pith.science/api/pith-number/ICVBGSN7ERMG6JCWT4X3K52KFH/graph.json","events_json":"https://pith.science/api/pith-number/ICVBGSN7ERMG6JCWT4X3K52KFH/events.json","paper":"https://pith.science/paper/ICVBGSN7"},"agent_actions":{"view_html":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH","download_json":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH.json","view_paper":"https://pith.science/paper/ICVBGSN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.05137&json=true","fetch_graph":"https://pith.science/api/pith-number/ICVBGSN7ERMG6JCWT4X3K52KFH/graph.json","fetch_events":"https://pith.science/api/pith-number/ICVBGSN7ERMG6JCWT4X3K52KFH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH/action/storage_attestation","attest_author":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH/action/author_attestation","sign_citation":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH/action/citation_signature","submit_replication":"https://pith.science/pith/ICVBGSN7ERMG6JCWT4X3K52KFH/action/replication_record"}},"created_at":"2026-07-05T09:45:28.955797+00:00","updated_at":"2026-07-05T09:45:28.955797+00:00"}