{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VDTQK5XSASNILLSSFYXYJDB7R6","short_pith_number":"pith:VDTQK5XS","schema_version":"1.0","canonical_sha256":"a8e70576f2049a85ae522e2f848c3f8f84c94674b2fbbba4cdb43ad8b29ae20b","source":{"kind":"arxiv","id":"2402.17168","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Data Science Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Kan Ren, Nan Chen, Qiyang Jiang, Xingyu Han, Yuge Zhang, Yuqing Yang","submitted_at":"2024-02-27T03:03:06Z","abstract_excerpt":"In the era of data-driven decision-making, the complexity of data analysis necessitates advanced expertise and tools of data science, presenting significant challenges even for specialists. Large Language Models (LLMs) have emerged as promising aids as data science agents, assisting humans in data analysis and processing. Yet their practical efficacy remains constrained by the varied demands of real-world applications and complicated analytical process. In this paper, we introduce DSEval -- a novel evaluation paradigm, as well as a series of innovative benchmarks tailored for assessing the per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17168","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-27T03:03:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fc8419328e14e12ca590ada9c3593db26e4cb8e3d1b3f2b4a2ea0edff3b635a9","abstract_canon_sha256":"63819eac09b1060a4754e8ba013c0c8a26c1f45f0125ba8a599f0778ac24973f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:43.503465Z","signature_b64":"DB46jZ7pj6e+nZ+FuW/udyKJnVyAAHmTMP/EpqhIl6YJogO0RLKUSUFhkbm6zWEylEYG0Q+htksK8H8gmOXQDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a8e70576f2049a85ae522e2f848c3f8f84c94674b2fbbba4cdb43ad8b29ae20b","last_reissued_at":"2026-07-05T07:49:43.502995Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:43.502995Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Data Science Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Kan Ren, Nan Chen, Qiyang Jiang, Xingyu Han, Yuge Zhang, Yuqing Yang","submitted_at":"2024-02-27T03:03:06Z","abstract_excerpt":"In the era of data-driven decision-making, the complexity of data analysis necessitates advanced expertise and tools of data science, presenting significant challenges even for specialists. Large Language Models (LLMs) have emerged as promising aids as data science agents, assisting humans in data analysis and processing. Yet their practical efficacy remains constrained by the varied demands of real-world applications and complicated analytical process. In this paper, we introduce DSEval -- a novel evaluation paradigm, as well as a series of innovative benchmarks tailored for assessing the per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17168","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17168","created_at":"2026-07-05T07:49:43.503090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17168v1","created_at":"2026-07-05T07:49:43.503090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17168","created_at":"2026-07-05T07:49:43.503090+00:00"},{"alias_kind":"pith_short_12","alias_value":"VDTQK5XSASNI","created_at":"2026-07-05T07:49:43.503090+00:00"},{"alias_kind":"pith_short_16","alias_value":"VDTQK5XSASNILLSS","created_at":"2026-07-05T07:49:43.503090+00:00"},{"alias_kind":"pith_short_8","alias_value":"VDTQK5XS","created_at":"2026-07-05T07:49:43.503090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00051","citing_title":"Business Utility of Large Language Models as Exploratory Data Analysis Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13564","citing_title":"Memory in the Age of AI Agents","ref_index":300,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6","json":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6.json","graph_json":"https://pith.science/api/pith-number/VDTQK5XSASNILLSSFYXYJDB7R6/graph.json","events_json":"https://pith.science/api/pith-number/VDTQK5XSASNILLSSFYXYJDB7R6/events.json","paper":"https://pith.science/paper/VDTQK5XS"},"agent_actions":{"view_html":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6","download_json":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6.json","view_paper":"https://pith.science/paper/VDTQK5XS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17168&json=true","fetch_graph":"https://pith.science/api/pith-number/VDTQK5XSASNILLSSFYXYJDB7R6/graph.json","fetch_events":"https://pith.science/api/pith-number/VDTQK5XSASNILLSSFYXYJDB7R6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6/action/storage_attestation","attest_author":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6/action/author_attestation","sign_citation":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6/action/citation_signature","submit_replication":"https://pith.science/pith/VDTQK5XSASNILLSSFYXYJDB7R6/action/replication_record"}},"created_at":"2026-07-05T07:49:43.503090+00:00","updated_at":"2026-07-05T07:49:43.503090+00:00"}