{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WSGE2ER45HFUT7OGIFHBK2GCGB","short_pith_number":"pith:WSGE2ER4","schema_version":"1.0","canonical_sha256":"b48c4d123ce9cb49fdc6414e1568c2307e6aa4e768da150a45c5525132da77cd","source":{"kind":"arxiv","id":"2201.10066","version":1},"attestation_state":"computed","paper":{"title":"Documenting Geographically and Contextually Diverse Data Sources: The BigScience Catalogue of Language Data and Resources","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DB"],"primary_cat":"cs.CL","authors_text":"Aitor Soroa, Alham Fikri Aji, Angelina McMillan-Major, Chris Emezue, Colin Leong, Daniel van Strien, Francesco De Toni, G\\'erard Dupont, Hady Elsahar, Kimbo Chen, Maraim Masoud, Nurulaqilla Khamis, Pedro Ortiz Suarez, Stella Biderman, Suzana Ili\\'c, Yacine Jernite, Zaid Alyafeai, Zeerak Talat","submitted_at":"2022-01-25T03:05:23Z","abstract_excerpt":"In recent years, large-scale data collection efforts have prioritized the amount of data collected in order to improve the modeling capabilities of large language models. This prioritization, however, has resulted in concerns with respect to the rights of data subjects represented in data collections, particularly when considering the difficulty in interrogating these collections due to insufficient documentation and tools for analysis. Mindful of these pitfalls, we present our methodology for a documentation-first, human-centered data collection project as part of the BigScience initiative. W"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.10066","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-01-25T03:05:23Z","cross_cats_sorted":["cs.DB"],"title_canon_sha256":"b9ceace9d77ad17bb1088155320497efb91b60f363cdf0910d46bfd79e351661","abstract_canon_sha256":"881f18f4e4f1d0bb006fa642081c095f3e5973509ad9777856570229e228d857"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:51:15.155738Z","signature_b64":"x1w75FJQ6UWHKzdj/Tt04xNfl+eqVCxDABAPPEtAWlUZCnBaKUVrjnvGbcCfS9++3I3+fKzuu4QzG3puddHdDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b48c4d123ce9cb49fdc6414e1568c2307e6aa4e768da150a45c5525132da77cd","last_reissued_at":"2026-07-05T03:51:15.155258Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:51:15.155258Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Documenting Geographically and Contextually Diverse Data Sources: The BigScience Catalogue of Language Data and Resources","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DB"],"primary_cat":"cs.CL","authors_text":"Aitor Soroa, Alham Fikri Aji, Angelina McMillan-Major, Chris Emezue, Colin Leong, Daniel van Strien, Francesco De Toni, G\\'erard Dupont, Hady Elsahar, Kimbo Chen, Maraim Masoud, Nurulaqilla Khamis, Pedro Ortiz Suarez, Stella Biderman, Suzana Ili\\'c, Yacine Jernite, Zaid Alyafeai, Zeerak Talat","submitted_at":"2022-01-25T03:05:23Z","abstract_excerpt":"In recent years, large-scale data collection efforts have prioritized the amount of data collected in order to improve the modeling capabilities of large language models. This prioritization, however, has resulted in concerns with respect to the rights of data subjects represented in data collections, particularly when considering the difficulty in interrogating these collections due to insufficient documentation and tools for analysis. Mindful of these pitfalls, we present our methodology for a documentation-first, human-centered data collection project as part of the BigScience initiative. W"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.10066","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.10066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.10066","created_at":"2026-07-05T03:51:15.155324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.10066v1","created_at":"2026-07-05T03:51:15.155324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.10066","created_at":"2026-07-05T03:51:15.155324+00:00"},{"alias_kind":"pith_short_12","alias_value":"WSGE2ER45HFU","created_at":"2026-07-05T03:51:15.155324+00:00"},{"alias_kind":"pith_short_16","alias_value":"WSGE2ER45HFUT7OG","created_at":"2026-07-05T03:51:15.155324+00:00"},{"alias_kind":"pith_short_8","alias_value":"WSGE2ER4","created_at":"2026-07-05T03:51:15.155324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":276,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":99,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB","json":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB.json","graph_json":"https://pith.science/api/pith-number/WSGE2ER45HFUT7OGIFHBK2GCGB/graph.json","events_json":"https://pith.science/api/pith-number/WSGE2ER45HFUT7OGIFHBK2GCGB/events.json","paper":"https://pith.science/paper/WSGE2ER4"},"agent_actions":{"view_html":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB","download_json":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB.json","view_paper":"https://pith.science/paper/WSGE2ER4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.10066&json=true","fetch_graph":"https://pith.science/api/pith-number/WSGE2ER45HFUT7OGIFHBK2GCGB/graph.json","fetch_events":"https://pith.science/api/pith-number/WSGE2ER45HFUT7OGIFHBK2GCGB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB/action/storage_attestation","attest_author":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB/action/author_attestation","sign_citation":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB/action/citation_signature","submit_replication":"https://pith.science/pith/WSGE2ER45HFUT7OGIFHBK2GCGB/action/replication_record"}},"created_at":"2026-07-05T03:51:15.155324+00:00","updated_at":"2026-07-05T03:51:15.155324+00:00"}