{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DMTQKLKY4OWDXOSIOPQPQTMR2Q","short_pith_number":"pith:DMTQKLKY","schema_version":"1.0","canonical_sha256":"1b27052d58e3ac3bba4873e0f84d91d433fbb4574356fa4365410ca83f8d6f23","source":{"kind":"arxiv","id":"2412.02764","version":2},"attestation_state":"computed","paper":{"title":"Drawing Pandas: A Benchmark for LLMs in Generating Plotting Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Egor Bogomolov, Sergey Titov, Timur Galimzyanov, Yaroslav Golubev","submitted_at":"2024-12-03T19:05:37Z","abstract_excerpt":"This paper introduces the human-curated PandasPlotBench dataset, designed to evaluate language models' effectiveness as assistants in visual data exploration. Our benchmark focuses on generating code for visualizing tabular data - such as a Pandas DataFrame - based on natural language instructions, complementing current evaluation tools and expanding their scope. The dataset includes 175 unique tasks. Our experiments assess several leading Large Language Models (LLMs) across three visualization libraries: Matplotlib, Seaborn, and Plotly. We show that the shortening of tasks has a minimal effec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.02764","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-12-03T19:05:37Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2a4b0bc78e940c3248cd7240356493512a5ba428d859c8c8541c9279440254da","abstract_canon_sha256":"9f71d7414b0a7ba9bc211a8a6c7d7e5f3bf518f55b792d959f4de3706090fd26"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:17.901957Z","signature_b64":"hEImiYuXXlUYjoKwPiE2vgkTxvN16nVDV6Mvd9wVyo3y5iI7bu0i9U1esS/20k2EkUSN5HfWR8OiYDyJb7hyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b27052d58e3ac3bba4873e0f84d91d433fbb4574356fa4365410ca83f8d6f23","last_reissued_at":"2026-07-05T10:20:17.901438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:17.901438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Drawing Pandas: A Benchmark for LLMs in Generating Plotting Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Egor Bogomolov, Sergey Titov, Timur Galimzyanov, Yaroslav Golubev","submitted_at":"2024-12-03T19:05:37Z","abstract_excerpt":"This paper introduces the human-curated PandasPlotBench dataset, designed to evaluate language models' effectiveness as assistants in visual data exploration. Our benchmark focuses on generating code for visualizing tabular data - such as a Pandas DataFrame - based on natural language instructions, complementing current evaluation tools and expanding their scope. The dataset includes 175 unique tasks. Our experiments assess several leading Large Language Models (LLMs) across three visualization libraries: Matplotlib, Seaborn, and Plotly. We show that the shortening of tasks has a minimal effec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.02764","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.02764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.02764","created_at":"2026-07-05T10:20:17.901500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.02764v2","created_at":"2026-07-05T10:20:17.901500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.02764","created_at":"2026-07-05T10:20:17.901500+00:00"},{"alias_kind":"pith_short_12","alias_value":"DMTQKLKY4OWD","created_at":"2026-07-05T10:20:17.901500+00:00"},{"alias_kind":"pith_short_16","alias_value":"DMTQKLKY4OWDXOSI","created_at":"2026-07-05T10:20:17.901500+00:00"},{"alias_kind":"pith_short_8","alias_value":"DMTQKLKY","created_at":"2026-07-05T10:20:17.901500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01883","citing_title":"PairCoder++: Pair Programming as a Universal Paradigm for Verified Code-Driven Multimodal and Structured-Artifact Generation","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q","json":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q.json","graph_json":"https://pith.science/api/pith-number/DMTQKLKY4OWDXOSIOPQPQTMR2Q/graph.json","events_json":"https://pith.science/api/pith-number/DMTQKLKY4OWDXOSIOPQPQTMR2Q/events.json","paper":"https://pith.science/paper/DMTQKLKY"},"agent_actions":{"view_html":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q","download_json":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q.json","view_paper":"https://pith.science/paper/DMTQKLKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.02764&json=true","fetch_graph":"https://pith.science/api/pith-number/DMTQKLKY4OWDXOSIOPQPQTMR2Q/graph.json","fetch_events":"https://pith.science/api/pith-number/DMTQKLKY4OWDXOSIOPQPQTMR2Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q/action/storage_attestation","attest_author":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q/action/author_attestation","sign_citation":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q/action/citation_signature","submit_replication":"https://pith.science/pith/DMTQKLKY4OWDXOSIOPQPQTMR2Q/action/replication_record"}},"created_at":"2026-07-05T10:20:17.901500+00:00","updated_at":"2026-07-05T10:20:17.901500+00:00"}