{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5IJQGLKPNHBYAHFDJDL5735FCY","short_pith_number":"pith:5IJQGLKP","schema_version":"1.0","canonical_sha256":"ea13032d4f69c3801ca348d7dfefa5163c2c19ba06bf0875fc815d735a7961f7","source":{"kind":"arxiv","id":"2107.02565","version":4},"attestation_state":"computed","paper":{"title":"Prioritized training on points that are learnable, worth learning, and not yet learned (workshop version)","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Adrien Morisot, Aidan N. Gomez, Andreas Kirsch, Jan Brauner, Mrinank Sharma, Muhammed Razzak, Sebastian Farquhar, S\\\"oren Mindermann, Winnie Xu, Yarin Gal","submitted_at":"2021-07-06T12:08:44Z","abstract_excerpt":"We introduce Goldilocks Selection, a technique for faster model training which selects a sequence of training points that are \"just right\". We propose an information-theoretic acquisition function -- the reducible validation loss -- and compute it with a small proxy model -- GoldiProx -- to efficiently choose training points that maximize information about a validation set. We show that the \"hard\" (e.g. high loss) points usually selected in the optimization literature are typically noisy, while the \"easy\" (e.g. low noise) samples often prioritized for curriculum learning confer less informatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.02565","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-06T12:08:44Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"42668613ab946bd77f112ecb3fa317712e25ba9e9a57188dcbe7745d5e24eb7c","abstract_canon_sha256":"9bdba3ebd28a75f03cdb0b0be156a69c852579081aca0c6e6a8208117878cdb7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:35.868569Z","signature_b64":"QX92bCTuv5cswf2TLMMpY5w8ue5/ypTm7+9IvZOM+Sv89ghz5XZbmis51LNQ3ilrsIfNhFzsmaO6iTPlsUg0Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea13032d4f69c3801ca348d7dfefa5163c2c19ba06bf0875fc815d735a7961f7","last_reissued_at":"2026-07-05T07:01:35.868138Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:35.868138Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prioritized training on points that are learnable, worth learning, and not yet learned (workshop version)","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Adrien Morisot, Aidan N. Gomez, Andreas Kirsch, Jan Brauner, Mrinank Sharma, Muhammed Razzak, Sebastian Farquhar, S\\\"oren Mindermann, Winnie Xu, Yarin Gal","submitted_at":"2021-07-06T12:08:44Z","abstract_excerpt":"We introduce Goldilocks Selection, a technique for faster model training which selects a sequence of training points that are \"just right\". We propose an information-theoretic acquisition function -- the reducible validation loss -- and compute it with a small proxy model -- GoldiProx -- to efficiently choose training points that maximize information about a validation set. We show that the \"hard\" (e.g. high loss) points usually selected in the optimization literature are typically noisy, while the \"easy\" (e.g. low noise) samples often prioritized for curriculum learning confer less informatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.02565","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.02565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.02565","created_at":"2026-07-05T07:01:35.868194+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.02565v4","created_at":"2026-07-05T07:01:35.868194+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.02565","created_at":"2026-07-05T07:01:35.868194+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IJQGLKPNHBY","created_at":"2026-07-05T07:01:35.868194+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IJQGLKPNHBYAHFD","created_at":"2026-07-05T07:01:35.868194+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IJQGLKP","created_at":"2026-07-05T07:01:35.868194+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.19361","citing_title":"Dynamic Skill Adaptation for Large Language Models","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY","json":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY.json","graph_json":"https://pith.science/api/pith-number/5IJQGLKPNHBYAHFDJDL5735FCY/graph.json","events_json":"https://pith.science/api/pith-number/5IJQGLKPNHBYAHFDJDL5735FCY/events.json","paper":"https://pith.science/paper/5IJQGLKP"},"agent_actions":{"view_html":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY","download_json":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY.json","view_paper":"https://pith.science/paper/5IJQGLKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.02565&json=true","fetch_graph":"https://pith.science/api/pith-number/5IJQGLKPNHBYAHFDJDL5735FCY/graph.json","fetch_events":"https://pith.science/api/pith-number/5IJQGLKPNHBYAHFDJDL5735FCY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY/action/storage_attestation","attest_author":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY/action/author_attestation","sign_citation":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY/action/citation_signature","submit_replication":"https://pith.science/pith/5IJQGLKPNHBYAHFDJDL5735FCY/action/replication_record"}},"created_at":"2026-07-05T07:01:35.868194+00:00","updated_at":"2026-07-05T07:01:35.868194+00:00"}