{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:QVXDM3TIZDV7IRRDK62WGSCICW","short_pith_number":"pith:QVXDM3TI","schema_version":"1.0","canonical_sha256":"856e366e68c8ebf4462357b563484815973d2fc6154a56629d337088a3da76b1","source":{"kind":"arxiv","id":"1702.01925","version":1},"attestation_state":"computed","paper":{"title":"Effects of Stop Words Elimination for Arabic Information Retrieval: A Comparative Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Ibrahim Abu El-khair","submitted_at":"2017-02-07T08:49:58Z","abstract_excerpt":"The effectiveness of three stop words lists for Arabic Information Retrieval---General Stoplist, Corpus-Based Stoplist, Combined Stoplist ---were investigated in this study. Three popular weighting schemes were examined: the inverse document frequency weight, probabilistic weighting, and statistical language modelling. The Idea is to combine the statistical approaches with linguistic approaches to reach an optimal performance, and compare their effect on retrieval. The LDC (Linguistic Data Consortium) Arabic Newswire data set was used with the Lemur Toolkit. The Best Match weighting scheme use"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1702.01925","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2017-02-07T08:49:58Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"1358afa52bd95a741202218995a6b7ca013a584578663d221def00343d199548","abstract_canon_sha256":"d47d8cab97272b28b5372c41068f6e7e90592eaff52bc28835f56316cbd6d18b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:51:10.854055Z","signature_b64":"8YM/G+XrqHXEb9hIghVMCjFrwbfMdg2QobLMhAvXVTFBNYcOZ6OH1+t3QjwjG8powhBY3IeAZINhfX/furCdCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"856e366e68c8ebf4462357b563484815973d2fc6154a56629d337088a3da76b1","last_reissued_at":"2026-05-18T00:51:10.853294Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:51:10.853294Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Effects of Stop Words Elimination for Arabic Information Retrieval: A Comparative Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Ibrahim Abu El-khair","submitted_at":"2017-02-07T08:49:58Z","abstract_excerpt":"The effectiveness of three stop words lists for Arabic Information Retrieval---General Stoplist, Corpus-Based Stoplist, Combined Stoplist ---were investigated in this study. Three popular weighting schemes were examined: the inverse document frequency weight, probabilistic weighting, and statistical language modelling. The Idea is to combine the statistical approaches with linguistic approaches to reach an optimal performance, and compare their effect on retrieval. The LDC (Linguistic Data Consortium) Arabic Newswire data set was used with the Lemur Toolkit. The Best Match weighting scheme use"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1702.01925","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1702.01925","created_at":"2026-05-18T00:51:10.853437+00:00"},{"alias_kind":"arxiv_version","alias_value":"1702.01925v1","created_at":"2026-05-18T00:51:10.853437+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1702.01925","created_at":"2026-05-18T00:51:10.853437+00:00"},{"alias_kind":"pith_short_12","alias_value":"QVXDM3TIZDV7","created_at":"2026-05-18T12:31:39.905425+00:00"},{"alias_kind":"pith_short_16","alias_value":"QVXDM3TIZDV7IRRD","created_at":"2026-05-18T12:31:39.905425+00:00"},{"alias_kind":"pith_short_8","alias_value":"QVXDM3TI","created_at":"2026-05-18T12:31:39.905425+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW","json":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW.json","graph_json":"https://pith.science/api/pith-number/QVXDM3TIZDV7IRRDK62WGSCICW/graph.json","events_json":"https://pith.science/api/pith-number/QVXDM3TIZDV7IRRDK62WGSCICW/events.json","paper":"https://pith.science/paper/QVXDM3TI"},"agent_actions":{"view_html":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW","download_json":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW.json","view_paper":"https://pith.science/paper/QVXDM3TI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1702.01925&json=true","fetch_graph":"https://pith.science/api/pith-number/QVXDM3TIZDV7IRRDK62WGSCICW/graph.json","fetch_events":"https://pith.science/api/pith-number/QVXDM3TIZDV7IRRDK62WGSCICW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW/action/storage_attestation","attest_author":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW/action/author_attestation","sign_citation":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW/action/citation_signature","submit_replication":"https://pith.science/pith/QVXDM3TIZDV7IRRDK62WGSCICW/action/replication_record"}},"created_at":"2026-05-18T00:51:10.853437+00:00","updated_at":"2026-05-18T00:51:10.853437+00:00"}