{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K7NH3OJL4VIVOGKIRNU26SZP3H","short_pith_number":"pith:K7NH3OJL","schema_version":"1.0","canonical_sha256":"57da7db92be5515719488b69af4b2fd9ef99df9bb73c1c717ed1a8e5e60b08a7","source":{"kind":"arxiv","id":"2304.09433","version":3},"attestation_state":"computed","paper":{"title":"Language Models Enable Simple Systems for Generating Structured Views of Heterogeneous Data Lakes","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrew Hojel, Avanika Narayan, Brandon Yang, Christopher R\\'e, Immanuel Trummer, Sabri Eyuboglu, Simran Arora","submitted_at":"2023-04-19T06:00:26Z","abstract_excerpt":"A long standing goal of the data management community is to develop general, automated systems that ingest semi-structured documents and output queryable tables without human effort or domain specific customization. Given the sheer variety of potential documents, state-of-the art systems make simplifying assumptions and use domain specific training. In this work, we ask whether we can maintain generality by using large language models (LLMs). LLMs, which are pretrained on broad data, can perform diverse downstream tasks simply conditioned on natural language task descriptions.\n  We propose and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.09433","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-19T06:00:26Z","cross_cats_sorted":[],"title_canon_sha256":"cc1a4b1d57d976500ecd33cc284d6ca05fd9c2d94e675c675fe4a6fe614f2b27","abstract_canon_sha256":"607f1dcabe2aa6f92751dea71c2e0d86861230be838deee7c074fb9263d05c10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:58.193142Z","signature_b64":"z5dPb5zyAlMclymnR1FaT64b/HPmMzwcpwTYTPD41dnyF9F6JtaMLOHm5JyeU9msLqpv/PVirzif5EwEL6FuCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57da7db92be5515719488b69af4b2fd9ef99df9bb73c1c717ed1a8e5e60b08a7","last_reissued_at":"2026-07-05T10:25:58.192660Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:58.192660Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Models Enable Simple Systems for Generating Structured Views of Heterogeneous Data Lakes","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrew Hojel, Avanika Narayan, Brandon Yang, Christopher R\\'e, Immanuel Trummer, Sabri Eyuboglu, Simran Arora","submitted_at":"2023-04-19T06:00:26Z","abstract_excerpt":"A long standing goal of the data management community is to develop general, automated systems that ingest semi-structured documents and output queryable tables without human effort or domain specific customization. Given the sheer variety of potential documents, state-of-the art systems make simplifying assumptions and use domain specific training. In this work, we ask whether we can maintain generality by using large language models (LLMs). LLMs, which are pretrained on broad data, can perform diverse downstream tasks simply conditioned on natural language task descriptions.\n  We propose and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.09433","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.09433/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.09433","created_at":"2026-07-05T10:25:58.192717+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.09433v3","created_at":"2026-07-05T10:25:58.192717+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.09433","created_at":"2026-07-05T10:25:58.192717+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7NH3OJL4VIV","created_at":"2026-07-05T10:25:58.192717+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7NH3OJL4VIVOGKI","created_at":"2026-07-05T10:25:58.192717+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7NH3OJL","created_at":"2026-07-05T10:25:58.192717+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19116","citing_title":"Towards an Agent-First Web: Redesigning the Web for AI Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11052","citing_title":"Attention Amnesia in Hybrid LLMs: When CoT Fine-Tuning Breaks Long-Range Recall, and How to Fix It","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30562","citing_title":"Morphing into Hybrid Attention Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16269","citing_title":"Train the Trainers -- An Agentic AI Framework for Peer-Based Mental Health Support in Battlefield Environments","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2312.06635","citing_title":"Gated Linear Attention Transformers with Hardware-Efficient Training","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02655","citing_title":"Semantic Data Processing with Holistic Data Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06464","citing_title":"Gated Delta Networks: Improving Mamba2 with Delta Rule","ref_index":295,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05838","citing_title":"MDN: Parallelizing Stepwise Momentum for Delta Linear Attention","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04749","citing_title":"AI Trust OS -- A Continuous Governance Framework for Autonomous AI Observability and Zero-Trust Compliance in Enterprise Environments","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05987","citing_title":"Flowr -- Scaling Up Retail Supply Chain Operations Through Agentic AI in Large Scale Supermarket Chains","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H","json":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H.json","graph_json":"https://pith.science/api/pith-number/K7NH3OJL4VIVOGKIRNU26SZP3H/graph.json","events_json":"https://pith.science/api/pith-number/K7NH3OJL4VIVOGKIRNU26SZP3H/events.json","paper":"https://pith.science/paper/K7NH3OJL"},"agent_actions":{"view_html":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H","download_json":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H.json","view_paper":"https://pith.science/paper/K7NH3OJL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.09433&json=true","fetch_graph":"https://pith.science/api/pith-number/K7NH3OJL4VIVOGKIRNU26SZP3H/graph.json","fetch_events":"https://pith.science/api/pith-number/K7NH3OJL4VIVOGKIRNU26SZP3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H/action/storage_attestation","attest_author":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H/action/author_attestation","sign_citation":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H/action/citation_signature","submit_replication":"https://pith.science/pith/K7NH3OJL4VIVOGKIRNU26SZP3H/action/replication_record"}},"created_at":"2026-07-05T10:25:58.192717+00:00","updated_at":"2026-07-05T10:25:58.192717+00:00"}