{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:S4SNQQXTQSSIRNGRCUKLI37R6E","short_pith_number":"pith:S4SNQQXT","schema_version":"1.0","canonical_sha256":"9724d842f384a488b4d11514b46ff1f123d4b125706ee563a5acfb3984f416c9","source":{"kind":"arxiv","id":"1910.14424","version":1},"attestation_state":"computed","paper":{"title":"Multi-Stage Document Ranking with BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Jimmy Lin, Kyunghyun Cho, Rodrigo Nogueira, Wei Yang","submitted_at":"2019-10-31T12:45:40Z","abstract_excerpt":"The advent of deep neural networks pre-trained via language modeling tasks has spurred a number of successful applications in natural language processing. This work explores one such popular model, BERT, in the context of document ranking. We propose two variants, called monoBERT and duoBERT, that formulate the ranking problem as pointwise and pairwise classification, respectively. These two models are arranged in a multi-stage ranking architecture to form an end-to-end search system. One major advantage of this design is the ability to trade off quality against latency by controlling the admi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.14424","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2019-10-31T12:45:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8a2aeabe0d5ccc64023d9c44aaae9ee122f7422d4f7a5198f1c7d23242a22f70","abstract_canon_sha256":"6856fdb540f61411c07719f13492d2f6afa0fa252cbc2f3f20b11accd5fba6c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:16:10.437781Z","signature_b64":"4b0kqgyK3kTZIfHYRVLC7O4I5PuWGQX1Fb6A+5fpJXoKN5pwgDsfaQGcBx3tDCQ5pQ/k7yit9vaW8Pjsm+eTAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9724d842f384a488b4d11514b46ff1f123d4b125706ee563a5acfb3984f416c9","last_reissued_at":"2026-07-05T00:16:10.437356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:16:10.437356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Stage Document Ranking with BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Jimmy Lin, Kyunghyun Cho, Rodrigo Nogueira, Wei Yang","submitted_at":"2019-10-31T12:45:40Z","abstract_excerpt":"The advent of deep neural networks pre-trained via language modeling tasks has spurred a number of successful applications in natural language processing. This work explores one such popular model, BERT, in the context of document ranking. We propose two variants, called monoBERT and duoBERT, that formulate the ranking problem as pointwise and pairwise classification, respectively. These two models are arranged in a multi-stage ranking architecture to form an end-to-end search system. One major advantage of this design is the ability to trade off quality against latency by controlling the admi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.14424","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.14424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.14424","created_at":"2026-07-05T00:16:10.437421+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.14424v1","created_at":"2026-07-05T00:16:10.437421+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.14424","created_at":"2026-07-05T00:16:10.437421+00:00"},{"alias_kind":"pith_short_12","alias_value":"S4SNQQXTQSSI","created_at":"2026-07-05T00:16:10.437421+00:00"},{"alias_kind":"pith_short_16","alias_value":"S4SNQQXTQSSIRNGR","created_at":"2026-07-05T00:16:10.437421+00:00"},{"alias_kind":"pith_short_8","alias_value":"S4SNQQXT","created_at":"2026-07-05T00:16:10.437421+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05968","citing_title":"InfluMatch: Frontier-Quality KOL Search at 4B-Model Cost","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04300","citing_title":"Argus-Retriever: Vision-LLM Late-Interaction Retrieval with Region-Aware Query-Conditioned MoE for Visual Document Retrieval","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28359","citing_title":"The Voronoi Bottleneck: Capacity-Aware Dense Retrieval for Product Search","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27559","citing_title":"A Sensitivity-Aware Test Collection for Search Among Personal Information","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2404.10981","citing_title":"A Survey on Retrieval-Augmented Text Generation for Large Language Models","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14751","citing_title":"Query pipeline optimization for cancer patient question answering systems","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.00709","citing_title":"RankFlow: A Multi-Role Collaborative Reranking Workflow Utilizing Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2312.02724","citing_title":"RankZephyr: Effective and Robust Zero-Shot Listwise Reranking is a Breeze!","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2201.10005","citing_title":"Text and Code Embeddings by Contrastive Pre-Training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11254","citing_title":"MIRA: An LLM-Assisted Benchmark for Multi-Category Integrated Retrieval","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27037","citing_title":"Hypencoder Revisited: Reproducibility and Analysis of Non-Linear Scoring for First-Stage Retrieval","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24608","citing_title":"Learning to Route Queries to Heads for Attention-based Re-ranking with Large Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03216","citing_title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22180","citing_title":"ResRank: Unifying Retrieval and Listwise Reranking via End-to-End Joint Training with Residual Passage Compression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17906","citing_title":"Bayesian Active Learning with Gaussian Processes Guided by LLM Relevance Scoring for Dense Passage Retrieval","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E","json":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E.json","graph_json":"https://pith.science/api/pith-number/S4SNQQXTQSSIRNGRCUKLI37R6E/graph.json","events_json":"https://pith.science/api/pith-number/S4SNQQXTQSSIRNGRCUKLI37R6E/events.json","paper":"https://pith.science/paper/S4SNQQXT"},"agent_actions":{"view_html":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E","download_json":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E.json","view_paper":"https://pith.science/paper/S4SNQQXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.14424&json=true","fetch_graph":"https://pith.science/api/pith-number/S4SNQQXTQSSIRNGRCUKLI37R6E/graph.json","fetch_events":"https://pith.science/api/pith-number/S4SNQQXTQSSIRNGRCUKLI37R6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E/action/storage_attestation","attest_author":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E/action/author_attestation","sign_citation":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E/action/citation_signature","submit_replication":"https://pith.science/pith/S4SNQQXTQSSIRNGRCUKLI37R6E/action/replication_record"}},"created_at":"2026-07-05T00:16:10.437421+00:00","updated_at":"2026-07-05T00:16:10.437421+00:00"}