{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:SAT4HCF4FN3V4OU5QCTBG4DVCP","short_pith_number":"pith:SAT4HCF4","schema_version":"1.0","canonical_sha256":"9027c388bc2b775e3a9d80a613707513cd750bfde352368ddc2fb492446df4d0","source":{"kind":"arxiv","id":"2608.11047","version":1},"attestation_state":"computed","paper":{"title":"V-FiLLM: Verified Financial LLM Reasoning Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CE","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alicia Larsen, Aulia Kharis Rakhamsari, Lara Turgut, Nino Antulov-Fantulin, Victoire Laurent","submitted_at":"2026-08-11T15:18:47Z","abstract_excerpt":"While existing benchmarks have made substantial progress in evaluating LLMs across STEM domains, financial reasoning over structured data remains comparatively less explored. We introduce V-FiLLM, a framework that generates financial reasoning benchmarks from executable computation trees grounded in real tables, yielding items whose answers are correct by construction. Trees are evaluated symbolically to obtain ground truth and rendered into natural-language questions, removing any model from the labeling loop, so items can be generated at arbitrary scale without annotation cost and without in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.11047","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-11T15:18:47Z","cross_cats_sorted":["cs.CE","cs.LG"],"title_canon_sha256":"f5266a7685120f9e072b0cbb6cd2e92e2fd897500bba638d5ad2c86cdf36d983","abstract_canon_sha256":"bea2e19f43def05b239a1a75ef50b924ed3695fc1d1fd0dc93bd2083ec8d90a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-12T01:24:29.681918Z","signature_b64":"iaCVXBOKjVvZdlAPWYDZfDljwCzWXFjs43Y6CNB8CNWEtHiyG6y5PdlXP4fTW8Z3VI/NhdYTzQBBALGmviVTCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9027c388bc2b775e3a9d80a613707513cd750bfde352368ddc2fb492446df4d0","last_reissued_at":"2026-08-12T01:24:29.680255Z","signature_status":"signed_v1","first_computed_at":"2026-08-12T01:24:29.680255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"V-FiLLM: Verified Financial LLM Reasoning Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CE","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alicia Larsen, Aulia Kharis Rakhamsari, Lara Turgut, Nino Antulov-Fantulin, Victoire Laurent","submitted_at":"2026-08-11T15:18:47Z","abstract_excerpt":"While existing benchmarks have made substantial progress in evaluating LLMs across STEM domains, financial reasoning over structured data remains comparatively less explored. We introduce V-FiLLM, a framework that generates financial reasoning benchmarks from executable computation trees grounded in real tables, yielding items whose answers are correct by construction. Trees are evaluated symbolically to obtain ground truth and rendered into natural-language questions, removing any model from the labeling loop, so items can be generated at arbitrary scale without annotation cost and without in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.11047","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.11047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.11047","created_at":"2026-08-12T01:24:29.684095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.11047v1","created_at":"2026-08-12T01:24:29.684095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.11047","created_at":"2026-08-12T01:24:29.684095+00:00"},{"alias_kind":"pith_short_12","alias_value":"SAT4HCF4FN3V","created_at":"2026-08-12T01:24:29.684095+00:00"},{"alias_kind":"pith_short_16","alias_value":"SAT4HCF4FN3V4OU5","created_at":"2026-08-12T01:24:29.684095+00:00"},{"alias_kind":"pith_short_8","alias_value":"SAT4HCF4","created_at":"2026-08-12T01:24:29.684095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP","json":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP.json","graph_json":"https://pith.science/api/pith-number/SAT4HCF4FN3V4OU5QCTBG4DVCP/graph.json","events_json":"https://pith.science/api/pith-number/SAT4HCF4FN3V4OU5QCTBG4DVCP/events.json","paper":"https://pith.science/paper/SAT4HCF4"},"agent_actions":{"view_html":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP","download_json":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP.json","view_paper":"https://pith.science/paper/SAT4HCF4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.11047&json=true","fetch_graph":"https://pith.science/api/pith-number/SAT4HCF4FN3V4OU5QCTBG4DVCP/graph.json","fetch_events":"https://pith.science/api/pith-number/SAT4HCF4FN3V4OU5QCTBG4DVCP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP/action/storage_attestation","attest_author":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP/action/author_attestation","sign_citation":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP/action/citation_signature","submit_replication":"https://pith.science/pith/SAT4HCF4FN3V4OU5QCTBG4DVCP/action/replication_record"}},"created_at":"2026-08-12T01:24:29.684095+00:00","updated_at":"2026-08-12T01:24:29.684095+00:00"}