{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QSPCKZ47AEAEQKDRTPFYCZLF4K","short_pith_number":"pith:QSPCKZ47","schema_version":"1.0","canonical_sha256":"849e25679f01004828719bcb816565e285bdacea387367d6c6f28b00f2d39750","source":{"kind":"arxiv","id":"2006.06894","version":1},"attestation_state":"computed","paper":{"title":"Google Dataset Search by the Numbers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB"],"primary_cat":"cs.IR","authors_text":"Natasha Noy, Omar Benjelloun, Shiyu Chen","submitted_at":"2020-06-12T00:54:15Z","abstract_excerpt":"Scientists, governments, and companies increasingly publish datasets on the Web. Google's Dataset Search extracts dataset metadata -- expressed using schema.org and similar vocabularies -- from Web pages in order to make datasets discoverable. Since we started the work on Dataset Search in 2016, the number of datasets described in schema.org has grown from about 500K to almost 30M. Thus, this corpus has become a valuable snapshot of data on the Web. To the best of our knowledge, this corpus is the largest and most diverse of its kind. We analyze this corpus and discuss where the datasets origi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.06894","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2020-06-12T00:54:15Z","cross_cats_sorted":["cs.DB"],"title_canon_sha256":"c09d10ef0bac27c769052c0e9e0c8c81849ef981cb0eec84fab985f543876125","abstract_canon_sha256":"6e8e94012a663c16b1ff49f5497824bb528a43b73a59b721b1d95ceab3386286"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:09:47.866880Z","signature_b64":"lvNKIZ5AWwqGNVQTl2Pat82lrdcKEjAZu+vlAEmcFf2R3nEX5pmpAZGNE33dpLjCtMuQbzYYolnEjInFbdIXDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"849e25679f01004828719bcb816565e285bdacea387367d6c6f28b00f2d39750","last_reissued_at":"2026-07-05T01:09:47.866486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:09:47.866486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Google Dataset Search by the Numbers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB"],"primary_cat":"cs.IR","authors_text":"Natasha Noy, Omar Benjelloun, Shiyu Chen","submitted_at":"2020-06-12T00:54:15Z","abstract_excerpt":"Scientists, governments, and companies increasingly publish datasets on the Web. Google's Dataset Search extracts dataset metadata -- expressed using schema.org and similar vocabularies -- from Web pages in order to make datasets discoverable. Since we started the work on Dataset Search in 2016, the number of datasets described in schema.org has grown from about 500K to almost 30M. Thus, this corpus has become a valuable snapshot of data on the Web. To the best of our knowledge, this corpus is the largest and most diverse of its kind. We analyze this corpus and discuss where the datasets origi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.06894","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.06894/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.06894","created_at":"2026-07-05T01:09:47.866547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.06894v1","created_at":"2026-07-05T01:09:47.866547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.06894","created_at":"2026-07-05T01:09:47.866547+00:00"},{"alias_kind":"pith_short_12","alias_value":"QSPCKZ47AEAE","created_at":"2026-07-05T01:09:47.866547+00:00"},{"alias_kind":"pith_short_16","alias_value":"QSPCKZ47AEAEQKDR","created_at":"2026-07-05T01:09:47.866547+00:00"},{"alias_kind":"pith_short_8","alias_value":"QSPCKZ47","created_at":"2026-07-05T01:09:47.866547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28787","citing_title":"Do Data Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K","json":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K.json","graph_json":"https://pith.science/api/pith-number/QSPCKZ47AEAEQKDRTPFYCZLF4K/graph.json","events_json":"https://pith.science/api/pith-number/QSPCKZ47AEAEQKDRTPFYCZLF4K/events.json","paper":"https://pith.science/paper/QSPCKZ47"},"agent_actions":{"view_html":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K","download_json":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K.json","view_paper":"https://pith.science/paper/QSPCKZ47","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.06894&json=true","fetch_graph":"https://pith.science/api/pith-number/QSPCKZ47AEAEQKDRTPFYCZLF4K/graph.json","fetch_events":"https://pith.science/api/pith-number/QSPCKZ47AEAEQKDRTPFYCZLF4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K/action/storage_attestation","attest_author":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K/action/author_attestation","sign_citation":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K/action/citation_signature","submit_replication":"https://pith.science/pith/QSPCKZ47AEAEQKDRTPFYCZLF4K/action/replication_record"}},"created_at":"2026-07-05T01:09:47.866547+00:00","updated_at":"2026-07-05T01:09:47.866547+00:00"}