{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:2KGSSM2PENC6ZPLVAMY7AYFKFP","short_pith_number":"pith:2KGSSM2P","schema_version":"1.0","canonical_sha256":"d28d29334f2345ecbd750331f060aa2bf0d1eaccb8ec8a8b1e9a405ba3f05b1f","source":{"kind":"arxiv","id":"2211.09961","version":1},"attestation_state":"computed","paper":{"title":"Path Independent Equilibrium Models Can Better Exploit Test-Time Computation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashwini Pokle, Cem Anil, Johannes Treutlein, Kaiqu Liang, Roger Grosse, Shaojie Bai, Yuhuai Wu, Zico Kolter","submitted_at":"2022-11-18T00:42:53Z","abstract_excerpt":"Designing networks capable of attaining better performance with an increased inference budget is important to facilitate generalization to harder problem instances. Recent efforts have shown promising results in this direction by making use of depth-wise recurrent networks. We show that a broad class of architectures named equilibrium models display strong upwards generalization, and find that stronger performance on harder examples (which require more iterations of inference to get correct) strongly correlates with the path independence of the system -- its tendency to converge to the same st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.09961","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-18T00:42:53Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"2afd9cb85b76ea08daa5c3d83f3b91cf0c787b4e33258b6696f7af4278393a51","abstract_canon_sha256":"b90c6fdd1e5853575f71f51aa13f7da12fae2ccd8658bc74b4458e5b68b73525"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:17:16.251233Z","signature_b64":"NaanmtcOdIxOQ6Yzj6MhQEmbugFSuWBZqv4qwniszIxtxFFWIXY66HL6dopy3H4PbiyFnhIyiTC7HmYSaasEAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d28d29334f2345ecbd750331f060aa2bf0d1eaccb8ec8a8b1e9a405ba3f05b1f","last_reissued_at":"2026-07-05T05:17:16.250805Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:17:16.250805Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Path Independent Equilibrium Models Can Better Exploit Test-Time Computation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashwini Pokle, Cem Anil, Johannes Treutlein, Kaiqu Liang, Roger Grosse, Shaojie Bai, Yuhuai Wu, Zico Kolter","submitted_at":"2022-11-18T00:42:53Z","abstract_excerpt":"Designing networks capable of attaining better performance with an increased inference budget is important to facilitate generalization to harder problem instances. Recent efforts have shown promising results in this direction by making use of depth-wise recurrent networks. We show that a broad class of architectures named equilibrium models display strong upwards generalization, and find that stronger performance on harder examples (which require more iterations of inference to get correct) strongly correlates with the path independence of the system -- its tendency to converge to the same st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.09961","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.09961/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.09961","created_at":"2026-07-05T05:17:16.250862+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.09961v1","created_at":"2026-07-05T05:17:16.250862+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09961","created_at":"2026-07-05T05:17:16.250862+00:00"},{"alias_kind":"pith_short_12","alias_value":"2KGSSM2PENC6","created_at":"2026-07-05T05:17:16.250862+00:00"},{"alias_kind":"pith_short_16","alias_value":"2KGSSM2PENC6ZPLV","created_at":"2026-07-05T05:17:16.250862+00:00"},{"alias_kind":"pith_short_8","alias_value":"2KGSSM2P","created_at":"2026-07-05T05:17:16.250862+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12946","citing_title":"Parcae: Scaling Laws For Stable Looped Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09168","citing_title":"ELT: Elastic Looped Transformers for Visual Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15259","citing_title":"Stability and Generalization in Looped Transformers","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP","json":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP.json","graph_json":"https://pith.science/api/pith-number/2KGSSM2PENC6ZPLVAMY7AYFKFP/graph.json","events_json":"https://pith.science/api/pith-number/2KGSSM2PENC6ZPLVAMY7AYFKFP/events.json","paper":"https://pith.science/paper/2KGSSM2P"},"agent_actions":{"view_html":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP","download_json":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP.json","view_paper":"https://pith.science/paper/2KGSSM2P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.09961&json=true","fetch_graph":"https://pith.science/api/pith-number/2KGSSM2PENC6ZPLVAMY7AYFKFP/graph.json","fetch_events":"https://pith.science/api/pith-number/2KGSSM2PENC6ZPLVAMY7AYFKFP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP/action/storage_attestation","attest_author":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP/action/author_attestation","sign_citation":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP/action/citation_signature","submit_replication":"https://pith.science/pith/2KGSSM2PENC6ZPLVAMY7AYFKFP/action/replication_record"}},"created_at":"2026-07-05T05:17:16.250862+00:00","updated_at":"2026-07-05T05:17:16.250862+00:00"}