{"as_of":"2026-08-19T15:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1aa0dd0ccea76177786761463e1e5a920f23c692ebfcdcf361eb6f5f835f563a","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T21:30:54.180653Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-25T05:55:10.325836Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T05:55:24.019027Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"cited_work":{"arxiv_id":"2602.20062","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.20062","snapshot_observed_at":"2026-07-01T02:17:19.837651Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","venue":null,"work_id":"4d8e5b33-fc2a-4f2a-ad75-d46f4b2aa37c","year":2026},"citing_paper":{"arxiv_id":"2605.20105","last_updated":"2026-05-19T16:56:56Z","snapshot_observed_at":"2026-08-14T10:15:44.943268Z","submitted_at":"2026-05-19T16:56:56Z","title":"Optimal Representation Size: High-Dimensional Analysis of Pretraining and Linear Probing","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T06:53:16.925588Z"},"links":{"cited_paper":"/paper/2602.20062","citing_paper":"/paper/2605.20105"},"observation_digest":"sha256:fcd15f9d9e59869754095ce6bde2d533b1afdb708441a03a6556e22704179c23","observation_id":"745444f6-ea4e-46ff-8ee8-50d251a35440","resolution":{"observed_at":"2026-07-01T02:17:19.837651Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"cited_work":{"arxiv_id":"2602.20062","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.20062","snapshot_observed_at":"2026-07-01T02:17:19.837651Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","venue":null,"work_id":"4d8e5b33-fc2a-4f2a-ad75-d46f4b2aa37c","year":2026},"citing_paper":{"arxiv_id":"2605.22972","last_updated":"2026-05-21T19:04:19Z","snapshot_observed_at":"2026-08-15T03:48:29.433293Z","submitted_at":"2026-05-21T19:04:19Z","title":"A mathematical theory of balancing relational generalization and memorization","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-25T05:55:10.325836Z"},"links":{"cited_paper":"/paper/2602.20062","citing_paper":"/paper/2605.22972"},"observation_digest":"sha256:bd787c059e05d170721bd84f34a83616bf44bea84108a102a3b9b152bd579d26","observation_id":"f93d44aa-e97b-45a6-8c3c-e82d51431e77","resolution":{"observed_at":"2026-07-01T02:17:19.837651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2602.20062/citation-record","integrity":"/paper/2602.20062/integrity","json":"/paper/2602.20062/citation-record.json","paper":"/paper/2602.20062"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.884441Z","title":"Neural networks as kernel learners: The silent alignment effect, 10 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.884441Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:20781256111d346c7695c6b0cc6637569a184b117de3439cc4a67ddf2cd9a887","observation_id":"59f0bb67-2507-4970-87db-f3892ad1695e","resolution":{"observed_at":"2026-08-02T21:30:49.884441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.916160Z","title":"M., Cholakkal, H., Shah, M., Yang, M.-H., and Khan, F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.916160Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:422975dc3e6adf6fb79ad22f9532f0c4dcb81f7afb55217cc4e20243eab98732","observation_id":"8d0ee721-d2a7-4e30-901a-9730a05b89ad","resolution":{"observed_at":"2026-08-02T21:30:49.916160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.987592Z","title":"S., Woodworth, B","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.987592Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:9cb21a4d6586d5604a1994dcc1f0c5161746f59d4ca50cafea3c2b985cefd907","observation_id":"3297e28c-2d48-4d67-b953-fdf13188f00a","resolution":{"observed_at":"2026-08-02T21:30:49.987592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.100342Z","title":"and Montanari, A","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.100342Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:dad1502d7aaf209285654ab316240acd4833839e661580d6105fce3abb8e3c7a","observation_id":"a16299bb-75e6-4c5b-9694-15f43c8e3e70","resolution":{"observed_at":"2026-08-02T21:30:50.100342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.209440Z","title":"and Montanari, A","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.209440Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:f92e08e9d927cb3429d6afc70d401033f6746200c1671d117f82ff501b124398","observation_id":"d868dc36-ebcd-4177-854a-f6cf7a26873c","resolution":{"observed_at":"2026-08-02T21:30:50.209440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.268733Z","title":"and M \\\"u ller, R","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.268733Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e007be32583634003065e843ee35a647add66c8bfb20c466189c9bee9934bf2c","observation_id":"c5a0ad6c-a66f-41b8-98f5-5e4d34c5beb6","resolution":{"observed_at":"2026-08-02T21:30:50.268733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.313876Z","title":"R., and Schulz-Baldes, H","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.313876Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ca5b9842e5691e5bf30288140122ad4fc37b88a354010a8069be53b78841a258","observation_id":"1a85d9e5-1126-4fbe-84b9-7ecf1f339ec9","resolution":{"observed_at":"2026-08-02T21:30:50.313876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.353814Z","title":"Incremental learning in diagonal linear networks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.353814Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:be55f52be23580269885d8e115d1088da1151ef9d1a1c2b02d3c0bf0f2d0e621","observation_id":"0f92376a-e77e-4f63-accc-5c40b8e580df","resolution":{"observed_at":"2026-08-02T21:30:50.353814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-02T21:30:50.405308Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.405308Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:256f39ab787b50fb9450808561ab36cfeb623da347c5eae8c67e4211e172331e","observation_id":"95dbc523-b5af-490f-8cd6-81973723dfed","resolution":{"observed_at":"2026-08-02T21:30:50.405308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.492722Z","title":"Exact learning dynamics of deep linear networks with prior knowledge","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.492722Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ed690ed14039f4c2483b3a72abb63f305a6722e6dc470434c49f4c293a3236e3","observation_id":"a9bd822d-ec60-4d37-873a-60dec6d26665","resolution":{"observed_at":"2026-08-02T21:30:50.492722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.594416Z","title":"and Bach, F","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.594416Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:b73b7cff05993e7a9335cc2454efa616deafa97cf0590cd8e93675f1ef25d32c","observation_id":"7a2baf06-380f-498c-b65c-b4dbbb39ae9e","resolution":{"observed_at":"2026-08-02T21:30:50.594416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.695889Z","title":"On lazy training in differentiable programming","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.695889Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:cb7d2f7cdac2dbbdb14390398a9e2734a5a5329f19a56ca58c3e70295b9f9146","observation_id":"3a2c5efa-99cf-474b-894c-aa759a367748","resolution":{"observed_at":"2026-08-02T21:30:50.695889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00194","last_updated":"2024-12-23T04:15:36Z","snapshot_observed_at":"2026-08-17T05:27:41.239892Z","submitted_at":"2024-02-29T23:46:28Z","title":"Ask Your Distribution Shift if Pre-Training is Right for You","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00194","snapshot_observed_at":"2026-08-02T21:30:50.791331Z","title":"Ask your distribution shift if pre-training is right for you","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.791331Z"},"links":{"cited_paper":"/paper/2403.00194","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:a4ddcf0190f18d946f11f3b777d53333d73f804a647eedcbfce24cdb58f0d7f7","observation_id":"d8a90f2f-58a3-470e-a3b1-6b9a35e89265","resolution":{"observed_at":"2026-08-02T21:30:50.791331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14623","last_updated":"2025-03-04T11:18:33Z","snapshot_observed_at":"2026-08-16T13:16:26.554379Z","submitted_at":"2024-09-22T23:19:04Z","title":"From Lazy to Rich: Exact Learning Dynamics in Deep Linear Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14623","snapshot_observed_at":"2026-08-02T21:30:50.893495Z","title":"C., Anguita, N., Proca, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.893495Z"},"links":{"cited_paper":"/paper/2409.14623","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c4912322c5d697c94cc86e17e10ca7f11d91fe5f239b1c958990fef6e0f4ea14","observation_id":"bb9fab7a-155a-40e6-b22b-bbfcdc86025d","resolution":{"observed_at":"2026-08-02T21:30:50.893495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.033052Z","title":null,"venue":null,"work_id":null,"year":1975},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.033052Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:cd8ba3881389146cd551bb4a0c9e3694542d70f772f5e7007ba77162f05365d0","observation_id":"006c9240-117a-46f7-b735-45e46987d7d1","resolution":{"observed_at":"2026-08-02T21:30:51.033052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.097585Z","title":"K., Paul, M., Kharaghani, S., Roy, D","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.097585Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4aa23a287d591b7dcd4f94767034279c00bf0eb8faa23d518d5dc8b28b4df254","observation_id":"76a65956-41f3-47e9-8680-0dae8e3b0dbf","resolution":{"observed_at":"2026-08-02T21:30:51.097585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.127385Z","title":"A theory of multineuronal dimensionality, dynamics and measurement","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.127385Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:88bf96f97ca9e3870aa719f87fca0e73cb48ce2388a7822da9f258c921b62726","observation_id":"449667df-ff04-4aab-ae3d-870c14bdd642","resolution":{"observed_at":"2026-08-02T21:30:51.127385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.197288Z","title":"R., and Aoi, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.197288Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:bb9cd3cd557e37f467b9e96d57c0199445aafc5dec69ee2ebc9b52ebb8857018","observation_id":"21a581be-8d00-40fb-94b9-e67b2ec89420","resolution":{"observed_at":"2026-08-02T21:30:51.197288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.243328Z","title":"Characterizing implicit bias in terms of optimization geometry","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.243328Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:20f1f4cbafeb24fb9f5a373e54ed973cd608616d37d49be9c7fe9cfa1a253a8d","observation_id":"a714be6b-bd1d-4a54-b47a-0eb60c3b47c2","resolution":{"observed_at":"2026-08-02T21:30:51.243328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.288371Z","title":"and Verd \\'u , S","venue":null,"work_id":null,"year":1983},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.288371Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:801438bfe9276deb888e24a3f7b26a0991d34642fa40a20bda18ac1c3dd4a815","observation_id":"c5c43e10-ea5f-4ffb-9dc0-33ff57426479","resolution":{"observed_at":"2026-08-02T21:30:51.288371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1608.08614","last_updated":"2016-12-10T13:37:06Z","snapshot_observed_at":"2026-08-14T21:42:06.055643Z","submitted_at":"2016-08-30T19:45:09Z","title":"What makes ImageNet good for transfer learning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1608.08614","snapshot_observed_at":"2026-08-02T21:30:51.355068Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.355068Z"},"links":{"cited_paper":"/paper/1608.08614","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c98ae64e638d8e36e0214549341936e686714dd9a4bc30ddce83af899562caaa","observation_id":"55c7ae16-7424-40ce-a31a-a65023293add","resolution":{"observed_at":"2026-08-02T21:30:51.355068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.404512Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.404512Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ffeefb89e237ab6d908cea06094093563a608e5f0d3ae1f1b6c20713ae9abdd4","observation_id":"38796d70-5e70-4226-9189-acd7406fa9f2","resolution":{"observed_at":"2026-08-02T21:30:51.404512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.439162Z","title":"Train on Validation (ToV): Fast data selection with applications to fine-tuning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.439162Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:716ba8c47c4f614d2d415b66f7b3b99f0355d014d188d1bc3da26526a2fb3daa","observation_id":"6e5ea2a4-2672-4f16-8bd4-25c9f88dae83","resolution":{"observed_at":"2026-08-02T21:30:51.439162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12786","last_updated":"2024-08-21T16:37:20Z","snapshot_observed_at":"2026-08-19T01:15:11.526401Z","submitted_at":"2023-11-21T18:51:04Z","title":"Mechanistically analyzing the effects of fine-tuning on procedurally defined tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12786","snapshot_observed_at":"2026-08-02T21:30:51.482772Z","title":"S., Dick, R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.482772Z"},"links":{"cited_paper":"/paper/2311.12786","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:00f9b1e0d9dc20ed5e06aa99bbe08a2e463e2c604b5c5e90c96608190a5c9917","observation_id":"222ab62a-eff7-478f-b07d-541fd0edfdff","resolution":{"observed_at":"2026-08-02T21:30:51.482772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02774","last_updated":"2024-05-05T00:08:00Z","snapshot_observed_at":"2026-08-19T12:32:37.273889Z","submitted_at":"2024-05-05T00:08:00Z","title":"Get more for less: Principled Data Selection for Warming Up Fine-Tuning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.02774","snapshot_observed_at":"2026-08-02T21:30:51.528872Z","title":"A., Sun, Y., Jahagirdar, H., Zhang, Y., Du, R., Sahu, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.528872Z"},"links":{"cited_paper":"/paper/2405.02774","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ce6b86581290cee9d95eeab7f87d2bf34c322dc424be640cf1b5ade33e0797e1","observation_id":"4dc8c455-ea4b-42ea-9e8c-799e53dfad81","resolution":{"observed_at":"2026-08-02T21:30:51.528872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.572721Z","title":"A., Xu, W., Avestimehr, A","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.572721Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:fbef8974518bc5519c818c5ed2176724135318443bfc787c7d82787b979c3747","observation_id":"dc8c2ec6-ab9b-4ab2-bf1e-6a43de119877","resolution":{"observed_at":"2026-08-02T21:30:51.572721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.614373Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.614373Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4b893bf7cb562c19cfa9d7a6f8186577cb2984178a4f6768f59fd95f7bad91b0","observation_id":"08a4bd9c-c927-490c-9ea0-52ee5443f3ed","resolution":{"observed_at":"2026-08-02T21:30:51.614373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.10054","last_updated":"2022-02-21T09:03:34Z","snapshot_observed_at":"2026-08-18T20:37:40.464812Z","submitted_at":"2022-02-21T09:03:34Z","title":"Fine-Tuning can Distort Pretrained Features and Underperform Out-of-Distribution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.10054","snapshot_observed_at":"2026-08-02T21:30:51.714613Z","title":"Fine-tuning can distort pretrained features and underperform out-of-distribution","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.714613Z"},"links":{"cited_paper":"/paper/2202.10054","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:789abceeda63e6fc12aceba3d7f72a54faec1cb0fae9d12a353f4634990ffdd7","observation_id":"22023ff0-97d8-4172-bf1d-f773f17e3571","resolution":{"observed_at":"2026-08-02T21:30:51.714613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06158","last_updated":"2024-10-12T21:38:28Z","snapshot_observed_at":"2026-08-18T15:27:34.066365Z","submitted_at":"2024-06-10T10:42:37Z","title":"Get rich quick: exact solutions reveal how unbalanced initializations promote rapid feature learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06158","snapshot_observed_at":"2026-08-02T21:30:51.868185Z","title":"Get rich quick: exact solutions reveal how unbalanced initializations promote rapid feature learning, 06 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.868185Z"},"links":{"cited_paper":"/paper/2406.06158","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:09624ce1fe3f989b3b07e8b5b5973245901bd0c6e9d9b33285497d08a1fd2b9a","observation_id":"be810198-8210-43b1-879f-11a02e250506","resolution":{"observed_at":"2026-08-02T21:30:51.868185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.10374","last_updated":"2019-01-04T22:16:58Z","snapshot_observed_at":"2026-08-16T20:36:12.569927Z","submitted_at":"2018-09-27T06:47:58Z","title":"An analytic theory of generalization dynamics and transfer learning in deep linear networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.10374","snapshot_observed_at":"2026-08-02T21:30:51.976081Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.976081Z"},"links":{"cited_paper":"/paper/1809.10374","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:81aa13984636a9834cfcbd007af82653b4c2e11b00aa634387ac6d8ffe1c4519","observation_id":"4def346e-c0ed-42d8-85a5-c6bde3e50cd3","resolution":{"observed_at":"2026-08-02T21:30:51.976081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.128369Z","title":"and Lindsey, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.128369Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:233718d2e0040a0304f39e1a1e067013269aa0cc97fb8fc129347b05937e1ed9","observation_id":"92933dcf-3120-44e7-9713-58a2446c998f","resolution":{"observed_at":"2026-08-02T21:30:52.128369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05890","last_updated":"2020-12-29T05:33:37Z","snapshot_observed_at":"2026-08-15T11:57:36.424287Z","submitted_at":"2019-06-13T18:52:00Z","title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.05890","snapshot_observed_at":"2026-08-02T21:30:52.266398Z","title":"and Li, J","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.266398Z"},"links":{"cited_paper":"/paper/1906.05890","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:bd0a56540853a7e70f3ecdb1753365f417c0666f25b30dd3f70b66ae3b9f8772","observation_id":"3f06f2f7-2c4f-4e83-8e0d-6896c75f72d3","resolution":{"observed_at":"2026-08-02T21:30:52.266398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.404458Z","title":"A kernel-based view of language model fine-tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.404458Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:232fe66835e274a29e3efcd46ee0e7a070b3854db5fb7ee99125af59eed9343e","observation_id":"9431c584-f450-432a-abe9-9a46a7809795","resolution":{"observed_at":"2026-08-02T21:30:52.404458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.571553Z","title":"Abide by the law and follow the flow: conservation laws for gradient flows, 12 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.571553Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:b2e1e36502f44bda67e4d9921d44255bc817a206c0eee11fe4b688b0e445ab12","observation_id":"d9c035d6-40c7-4937-bdd4-7cd8522e6607","resolution":{"observed_at":"2026-08-02T21:30:52.571553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.693140Z","title":null,"venue":null,"work_id":null,"year":1987},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.693140Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c18af6599adf6447ce4ef8c12d169e2ae24fd20bd1c8dfac1e1ff6ab45310b68","observation_id":"0c29b522-a8f3-4d71-8ca1-f5a0f933ceb3","resolution":{"observed_at":"2026-08-02T21:30:52.693140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1310.5479","last_updated":"2013-10-21T09:42:02Z","snapshot_observed_at":"2026-08-14T23:58:16.984302Z","submitted_at":"2013-10-21T09:42:02Z","title":"Applications of Large Random Matrices in Communications Engineering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1310.5479","snapshot_observed_at":"2026-08-02T21:30:52.805911Z","title":"R., Alfano, G., Zaidel, B","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.805911Z"},"links":{"cited_paper":"/paper/1310.5479","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:a14329f25d96a4ce2a5bf52ec5c342c8373b97c06575675f791e6721cd9edb15","observation_id":"93f1a70e-722c-4fa2-acf2-d6f997f13faf","resolution":{"observed_at":"2026-08-02T21:30:52.805911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.899019Z","title":"S., Gunasekar, S., Lee, J., Srebro, N., and Soudry, D","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.899019Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:699c51cf2d4e9782aa583cf279f6b5bd44212a07cd388402ed221509995819a9","observation_id":"bde92dd5-183e-47a9-8a75-8ee7c50c5330","resolution":{"observed_at":"2026-08-02T21:30:52.899019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.975413Z","title":"S., Ravichandran, K., Srebro, N., and Soudry, D","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.975413Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:8658404b7e6fd5e10d8e3e90470044d6ae45ce28f1eb7f48c66f9ce964be9dd8","observation_id":"c22e3cac-db5c-47f0-9256-e3f8d218b7ab","resolution":{"observed_at":"2026-08-02T21:30:52.975413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13296","last_updated":"2024-10-30T01:04:15Z","snapshot_observed_at":"2026-08-16T13:24:19.028622Z","submitted_at":"2024-08-23T14:48:02Z","title":"The Ultimate Guide to Fine-Tuning LLMs from Basics to Breakthroughs: An Exhaustive Review of Technologies, Research, Best Practices, Applied Research Challenges and Opportunities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13296","snapshot_observed_at":"2026-08-02T21:30:53.064503Z","title":"B., Zafar, A., Khan, A., and Shahid, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.064503Z"},"links":{"cited_paper":"/paper/2408.13296","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ebeec093e5acb9759f27a46e7084825a6c84d20de4b016d68555b8e295f35721","observation_id":"f21c03c0-66a1-4bba-9af9-ef00efa212d4","resolution":{"observed_at":"2026-08-02T21:30:53.064503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.136876Z","title":"and Flammarion, N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.136876Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:73247bb9f048d769d505604b9bb012880da3de076dde4fa476d0bcac84f4bd51","observation_id":"d851012b-3a44-4755-92f1-c5e1f9d2b4b1","resolution":{"observed_at":"2026-08-02T21:30:53.136876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.229231Z","title":"Implicit bias of sgd for diagonal linear networks: a provable benefit of stochasticity","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.229231Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:21edf7bf46a859ccab39591136130affc700de75aeecd37d272d98f7d363c0e1","observation_id":"d819b0ae-55b1-4cc3-a180-33aa0570a2d3","resolution":{"observed_at":"2026-08-02T21:30:53.229231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.323905Z","title":null,"venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.323905Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:3c5266138bad86a7a9c4a7e59550d861d5a6eab771e9111abb76713aad95dab7","observation_id":"42bd3e56-58a5-40ed-96d2-94e6a7334b4c","resolution":{"observed_at":"2026-08-02T21:30:53.323905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.386758Z","title":"How do infinite width bounded norm networks look in function space? In Conference on Learning Theory, pp.\\ 2667--2690","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.386758Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:1471bb412b2470badd1954d621efe1be386f10df6a726ee6e643c1628337b214","observation_id":"af03399a-9ae5-4531-9cc8-f0deb651bc0d","resolution":{"observed_at":"2026-08-02T21:30:53.386758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.456359Z","title":"L., and Ganguli, S","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.456359Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4d1abfe6876c45fadf0e025f469a0e9c4c35fb13e2be3286fe06378f0f5727bc","observation_id":"21bbf31b-07a8-4f02-a7f4-ae55c7ab8873","resolution":{"observed_at":"2026-08-02T21:30:53.456359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.551434Z","title":"M., McClelland, J","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.551434Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e913723d8bac40a2784ce81d0d99173f8f1f8cc07ec257556baf44523828a39f","observation_id":"02b0c49d-8e12-4dd6-a58d-8edafaa4affe","resolution":{"observed_at":"2026-08-02T21:30:53.551434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.658605Z","title":"A theoretical analysis of fine-tuning with linear teachers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.658605Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:491d84fbc7d5e864c8cbae4dbb032234570f5de41912c8118f47fb1117c77232","observation_id":"d87e80e0-6f31-4e46-8bbc-5cb87865e881","resolution":{"observed_at":"2026-08-02T21:30:53.658605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.766153Z","title":"S., Gunasekar, S., and Srebro, N","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.766153Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:f17aa55f8a283ccedad6a6c087df8b7f23cf8698d19827a48b6da49b00026103","observation_id":"b9f7f493-3d3b-4472-ba5c-e4d11c9c1d5f","resolution":{"observed_at":"2026-08-02T21:30:53.766153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08194","last_updated":"2025-07-07T20:57:29Z","snapshot_observed_at":"2026-08-18T19:38:37.069359Z","submitted_at":"2024-10-10T17:58:26Z","title":"Features are fate: a theory of transfer learning in high-dimensional regression","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08194","snapshot_observed_at":"2026-08-02T21:30:53.816520Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.816520Z"},"links":{"cited_paper":"/paper/2410.08194","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e78398f7789f1f4f6edea315b0cee0026c5d725f8fde2bd5e62e5f8dd94a59ca","observation_id":"1b790542-f1a9-42bf-bbc1-41d2f62b412d","resolution":{"observed_at":"2026-08-02T21:30:53.816520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.872616Z","title":"and Sato, I","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.872616Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:1fcf69cb5910958d8780c2b1ab50f5f201568d8ec53b337c19076af254fbf247","observation_id":"9a2fa254-11d8-406b-8bd3-4bc99f466981","resolution":{"observed_at":"2026-08-02T21:30:53.872616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.922347Z","title":"and Lu, W","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.922347Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:95accc5942fa4817d3f2762313feec8aa7c4566cef69e71f2f61c58ec17233b1","observation_id":"17d811cc-42a8-4554-9269-ea3569dda412","resolution":{"observed_at":"2026-08-02T21:30:53.922347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10012","last_updated":"2022-06-20T21:23:28Z","snapshot_observed_at":"2026-08-16T16:51:39.151684Z","submitted_at":"2022-06-20T21:23:28Z","title":"Limitations of the NTK for Understanding Generalization in Deep Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10012","snapshot_observed_at":"2026-08-02T21:30:53.974565Z","title":"Limitations of the ntk for understanding generalization in deep learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.974565Z"},"links":{"cited_paper":"/paper/2206.10012","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:6cbcfa4124efb10296dce3751eeebfcb9c53baa38ecfd240408798b81200bf1f","observation_id":"e20161cf-43e6-4be2-8c21-a002f48ac39d","resolution":{"observed_at":"2026-08-02T21:30:53.974565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.028680Z","title":"D., Moroshko, E., Savarese, P., Golan, I., Soudry, D., and Srebro, N","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.028680Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:9eda504eebe10b7c32e2cc9d06673123b28875791063ebed61f13150edfef469","observation_id":"2a1577c9-56bd-489a-9557-235a30a3ba19","resolution":{"observed_at":"2026-08-02T21:30:54.028680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.086358Z","title":"How transferable are features in deep neural networks? Advances in neural information processing systems, 27, 2014","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.086358Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:5ecaaf019d7578916dad69f15beec7bb0899ee3786c5c600726fce35f3bb7ce1","observation_id":"dbc813df-f6c7-4054-a144-af0f6cdfda0a","resolution":{"observed_at":"2026-08-02T21:30:54.086358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1611.03530","last_updated":"2017-02-26T19:36:40Z","snapshot_observed_at":"2026-08-01T16:56:59.989486Z","submitted_at":"2016-11-10T22:02:36Z","title":"Understanding deep learning requires rethinking generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.03530","snapshot_observed_at":"2026-08-02T21:30:54.126249Z","title":"Understanding deep learning requires rethinking generalization","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.126249Z"},"links":{"cited_paper":"/paper/1611.03530","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:dc28eaf77afd5449cf2c1a3bef250f678b7c72e39d4c6dbb4ee9e26554f79657","observation_id":"e137aa50-0b28-4426-a65f-3f5aec7f2996","resolution":{"observed_at":"2026-08-02T21:30:54.126249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.180653Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.180653Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:81a0eac3985519e7c64f2a398e50f9ca1d85c7a68f0ea5966ce2b9e40513990d","observation_id":"37204e22-915a-4584-a8e6-0c62d39a36f1","resolution":{"observed_at":"2026-08-02T21:30:54.180653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-13T13:11:34.685621Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":55,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 2 inbound Pith citation observations for arXiv:2602.20062."}