{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IOMUOLOVLEAQ5C7XDMHMMC75LU","short_pith_number":"pith:IOMUOLOV","schema_version":"1.0","canonical_sha256":"4399472dd559010e8bf71b0ec60bfd5d0ca8d5cb7e2a3065a41522780da822cb","source":{"kind":"arxiv","id":"2410.01774","version":2},"attestation_state":"computed","paper":{"title":"Trained Transformer Classifiers Generalize and Exhibit Benign Overfitting In-Context","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Gal Vardi, Spencer Frei","submitted_at":"2024-10-02T17:30:21Z","abstract_excerpt":"Transformers have the capacity to act as supervised learning algorithms: by properly encoding a set of labeled training (\"in-context\") examples and an unlabeled test example into an input sequence of vectors of the same dimension, the forward pass of the transformer can produce predictions for that unlabeled test example. A line of recent work has shown that when linear transformers are pre-trained on random instances for linear regression tasks, these trained transformers make predictions using an algorithm similar to that of ordinary least squares. In this work, we investigate the behavior o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01774","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T17:30:21Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"2c642f83022352a953c06b895ae03098298e88bede64ae61182ceaffad709e71","abstract_canon_sha256":"0a32c51fe1b63dd4d03aabd0f05ad21f40b916f1bbe0a34cb0de2be6939a1d6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:36.418769Z","signature_b64":"TCglobHRtXeWB+dwD9saMQDbN6QUtfVGD6YzFB3kGdQqIRSMcdo1p4KXq9K7rKZisZYngoZ1glXj4oQRzUfIAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4399472dd559010e8bf71b0ec60bfd5d0ca8d5cb7e2a3065a41522780da822cb","last_reissued_at":"2026-07-05T09:48:36.418333Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:36.418333Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Trained Transformer Classifiers Generalize and Exhibit Benign Overfitting In-Context","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Gal Vardi, Spencer Frei","submitted_at":"2024-10-02T17:30:21Z","abstract_excerpt":"Transformers have the capacity to act as supervised learning algorithms: by properly encoding a set of labeled training (\"in-context\") examples and an unlabeled test example into an input sequence of vectors of the same dimension, the forward pass of the transformer can produce predictions for that unlabeled test example. A line of recent work has shown that when linear transformers are pre-trained on random instances for linear regression tasks, these trained transformers make predictions using an algorithm similar to that of ordinary least squares. In this work, we investigate the behavior o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01774","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01774/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01774","created_at":"2026-07-05T09:48:36.418389+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01774v2","created_at":"2026-07-05T09:48:36.418389+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01774","created_at":"2026-07-05T09:48:36.418389+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOMUOLOVLEAQ","created_at":"2026-07-05T09:48:36.418389+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOMUOLOVLEAQ5C7X","created_at":"2026-07-05T09:48:36.418389+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOMUOLOV","created_at":"2026-07-05T09:48:36.418389+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06314","citing_title":"When Does $\\ell_2$-Boosting Overfit Benignly? High-Dimensional Risk Asymptotics and the $\\ell_1$ Implicit Bias","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06314","citing_title":"When Does $\\ell_2$-Boosting Overfit Benignly? High-Dimensional Risk Asymptotics and the $\\ell_1$ Implicit Bias","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19724","citing_title":"Benign Overfitting in Adversarial Training for Vision Transformers","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU","json":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU.json","graph_json":"https://pith.science/api/pith-number/IOMUOLOVLEAQ5C7XDMHMMC75LU/graph.json","events_json":"https://pith.science/api/pith-number/IOMUOLOVLEAQ5C7XDMHMMC75LU/events.json","paper":"https://pith.science/paper/IOMUOLOV"},"agent_actions":{"view_html":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU","download_json":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU.json","view_paper":"https://pith.science/paper/IOMUOLOV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01774&json=true","fetch_graph":"https://pith.science/api/pith-number/IOMUOLOVLEAQ5C7XDMHMMC75LU/graph.json","fetch_events":"https://pith.science/api/pith-number/IOMUOLOVLEAQ5C7XDMHMMC75LU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU/action/storage_attestation","attest_author":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU/action/author_attestation","sign_citation":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU/action/citation_signature","submit_replication":"https://pith.science/pith/IOMUOLOVLEAQ5C7XDMHMMC75LU/action/replication_record"}},"created_at":"2026-07-05T09:48:36.418389+00:00","updated_at":"2026-07-05T09:48:36.418389+00:00"}