{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GSN5P2O7SMDB6R6EZRXCLSQFWW","short_pith_number":"pith:GSN5P2O7","schema_version":"1.0","canonical_sha256":"349bd7e9df93061f47c4cc6e25ca05b5bd6ac9cd56d03f295070fc40b83b2ed3","source":{"kind":"arxiv","id":"2407.03618","version":1},"attestation_state":"computed","paper":{"title":"BM25S: Orders of magnitude faster lexical search via eager sparse scoring","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Xing Han L\\`u","submitted_at":"2024-07-04T04:01:05Z","abstract_excerpt":"We introduce BM25S, an efficient Python-based implementation of BM25 that only depends on Numpy and Scipy. BM25S achieves up to a 500x speedup compared to the most popular Python-based framework by eagerly computing BM25 scores during indexing and storing them into sparse matrices. It also achieves considerable speedups compared to highly optimized Java-based implementations, which are used by popular commercial products. Finally, BM25S reproduces the exact implementation of five BM25 variants based on Kamphuis et al. (2020) by extending eager scoring to non-sparse variants using a novel score"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03618","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-07-04T04:01:05Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ee4581acccd553121714cd91c15b1765abdd067ac3671928ad19c35b57823511","abstract_canon_sha256":"20c05bb3fa201d1d702521ede79354546f2552245f991d687b6332e724b65a28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:57.520639Z","signature_b64":"EFTmVmUlEJzRuZYWXx9QYK0VtjjwaAhAgugVciaQtsqBk15KEeWmtHrWW1wCjZE1XSjeK0NBblI6XA1N1rAHCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"349bd7e9df93061f47c4cc6e25ca05b5bd6ac9cd56d03f295070fc40b83b2ed3","last_reissued_at":"2026-07-05T08:39:57.520163Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:57.520163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BM25S: Orders of magnitude faster lexical search via eager sparse scoring","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Xing Han L\\`u","submitted_at":"2024-07-04T04:01:05Z","abstract_excerpt":"We introduce BM25S, an efficient Python-based implementation of BM25 that only depends on Numpy and Scipy. BM25S achieves up to a 500x speedup compared to the most popular Python-based framework by eagerly computing BM25 scores during indexing and storing them into sparse matrices. It also achieves considerable speedups compared to highly optimized Java-based implementations, which are used by popular commercial products. Finally, BM25S reproduces the exact implementation of five BM25 variants based on Kamphuis et al. (2020) by extending eager scoring to non-sparse variants using a novel score"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03618","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03618/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03618","created_at":"2026-07-05T08:39:57.520215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03618v1","created_at":"2026-07-05T08:39:57.520215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03618","created_at":"2026-07-05T08:39:57.520215+00:00"},{"alias_kind":"pith_short_12","alias_value":"GSN5P2O7SMDB","created_at":"2026-07-05T08:39:57.520215+00:00"},{"alias_kind":"pith_short_16","alias_value":"GSN5P2O7SMDB6R6E","created_at":"2026-07-05T08:39:57.520215+00:00"},{"alias_kind":"pith_short_8","alias_value":"GSN5P2O7","created_at":"2026-07-05T08:39:57.520215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05927","citing_title":"CMDR: Contextual Multimodal Document Retrieval","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18307","citing_title":"DRIFT: Refining Instruction Data via On-Policy Data Attribution","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17255","citing_title":"MLLP-VRAIN UPV system for the IWSLT 2026 Simultaneous Speech Translation task","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06880","citing_title":"Towards Retrieving Interaction Spaces for Agentic Search","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04704","citing_title":"Extraction and Search in Rocq: Theorems, Definitions and Their dependencies","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14348","citing_title":"Legal Retrieval for Public Defenders","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20537","citing_title":"What Do Biomedical NER and Entity Linking Benchmarks Measure? A Corpus-Centric Diagnostic Framework","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18561","citing_title":"Improving BM25 Code Retrieval Under Fixed Generic Tokenization: Adaptive q-Log Odds as a Drop-In BM25 Fix","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.11541","citing_title":"Question-Adaptive Graph Learning for Multi-hop Retrieval Augmented Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29002","citing_title":"Understand and Accelerate Memory Processing Pipeline for Large Language Model Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12477","citing_title":"MEME: Multi-entity & Evolving Memory Evaluation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12313","citing_title":"Overview of the MedHopQA track at BioCreative IX: track description, participation and evaluation of systems for multi-hop medical question answering","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12028","citing_title":"Caraman at SemEval-2026 Task 8: Three-Stage Multi-Turn Retrieval with Query Rewriting, Hybrid Search, and Cross-Encoder Reranking","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01399","citing_title":"Verbal-R3: Verbal Reranker as the Missing Bridge between Retrieval and Reasoning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00529","citing_title":"Hierarchical Abstract Tree for Cross-Document Retrieval-Augmented Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06006","citing_title":"From Articles to Premises: Building PrimeFacts, an Extraction Methodology and Resource for Fact-Checking Evidence","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04763","citing_title":"How Does Chunking Affect Retrieval-Augmented Code Completion? A Controlled Empirical Study","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW","json":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW.json","graph_json":"https://pith.science/api/pith-number/GSN5P2O7SMDB6R6EZRXCLSQFWW/graph.json","events_json":"https://pith.science/api/pith-number/GSN5P2O7SMDB6R6EZRXCLSQFWW/events.json","paper":"https://pith.science/paper/GSN5P2O7"},"agent_actions":{"view_html":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW","download_json":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW.json","view_paper":"https://pith.science/paper/GSN5P2O7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03618&json=true","fetch_graph":"https://pith.science/api/pith-number/GSN5P2O7SMDB6R6EZRXCLSQFWW/graph.json","fetch_events":"https://pith.science/api/pith-number/GSN5P2O7SMDB6R6EZRXCLSQFWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW/action/storage_attestation","attest_author":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW/action/author_attestation","sign_citation":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW/action/citation_signature","submit_replication":"https://pith.science/pith/GSN5P2O7SMDB6R6EZRXCLSQFWW/action/replication_record"}},"created_at":"2026-07-05T08:39:57.520215+00:00","updated_at":"2026-07-05T08:39:57.520215+00:00"}