{"as_of":"2026-08-09T22:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:de7dffe972ae66527adb7972a1b7b49e332fc9a50b032dc1cd36b18e9a2de105","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:40:35.806185Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T12:06:59.819223Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:09:41.618566Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"cited_work":{"arxiv_id":"2506.01901","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01901","snapshot_observed_at":"2026-07-04T08:09:41.618566Z","title":"Understanding overadaptation in supervised fine- tuning: The role of ensemble methods.arXiv preprint arXiv:2506.01901","venue":null,"work_id":"c17e6cb7-3892-4081-bf50-2e077f9bb578","year":2025},"citing_paper":{"arxiv_id":"2605.01954","last_updated":"2026-05-03T16:37:52Z","snapshot_observed_at":"2026-07-06T23:15:07.159340Z","submitted_at":"2026-05-03T16:37:52Z","title":"Moira: Language-driven Hierarchical Reinforcement Learning for Pair Trading","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T17:08:46.405278Z"},"links":{"cited_paper":"/paper/2506.01901","citing_paper":"/paper/2605.01954"},"observation_digest":"sha256:50bf81a7287ca2f86b865f4d30f31103946777e7d088e04dc367db28fad25f94","observation_id":"0aa9d925-374b-4faa-9840-c335b232bdbf","resolution":{"observed_at":"2026-05-11T16:26:05.994115Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"cited_work":{"arxiv_id":"2506.01901","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01901","snapshot_observed_at":"2026-07-04T08:09:41.618566Z","title":"Understanding overadaptation in supervised fine- tuning: The role of ensemble methods.arXiv preprint arXiv:2506.01901","venue":null,"work_id":"c17e6cb7-3892-4081-bf50-2e077f9bb578","year":2025},"citing_paper":{"arxiv_id":"2605.22126","last_updated":"2026-05-21T08:00:49Z","snapshot_observed_at":"2026-07-06T23:32:30.578213Z","submitted_at":"2026-05-21T08:00:49Z","title":"AesFormer: Transform Everyday Photos into Beautiful Memories","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T07:09:40.792161Z"},"links":{"cited_paper":"/paper/2506.01901","citing_paper":"/paper/2605.22126"},"observation_digest":"sha256:1f5da14e11fe1303848d9a9398dd98070f475c8620f5b3ea456865c0b60ae4b7","observation_id":"6e1c9936-a366-4a10-b130-ac2da7024a4a","resolution":{"observed_at":"2026-05-22T07:11:12.673045Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"cited_work":{"arxiv_id":"2506.01901","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01901","snapshot_observed_at":"2026-07-04T08:09:41.618566Z","title":"Understanding overadaptation in supervised fine- tuning: The role of ensemble methods.arXiv preprint arXiv:2506.01901","venue":null,"work_id":"c17e6cb7-3892-4081-bf50-2e077f9bb578","year":2025},"citing_paper":{"arxiv_id":"2606.21890","last_updated":"2026-06-20T05:39:59Z","snapshot_observed_at":"2026-08-04T01:30:26.373415Z","submitted_at":"2026-06-20T05:39:59Z","title":"Scaling Performance and Low-Resource Annotation with Many-Shot In-Context Learning for Named Entity Recognition","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-26T12:06:59.819223Z"},"links":{"cited_paper":"/paper/2506.01901","citing_paper":"/paper/2606.21890"},"observation_digest":"sha256:6fae5a3f7d49e9578896190c051551442c0b035dfc33c991d7cac9350270d9eb","observation_id":"769fb3fd-cfc6-4b27-a6bd-779b8fcbf3e0","resolution":{"observed_at":"2026-07-04T08:09:41.620182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.01901/citation-record","integrity":"/paper/2506.01901/integrity","json":"/paper/2506.01901/citation-record.json","paper":"/paper/2506.01901"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:48.808358Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:48.808358Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:fb496703a9fbd9dffc77868f8a3e69aed3c52d82f65fbf3ea501b4535e17e2e4","observation_id":"1e2bfb6e-bbb7-4c6a-b470-40654c64c5ae","resolution":{"observed_at":"2026-08-07T11:39:48.808358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T11:39:49.596584Z","title":"L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:49.596584Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:aaf9cde69baec7543e5ab427545902291e70029aa2d24fa4d2c0dfb1b8f65a86","observation_id":"969956b6-1460-48f4-8a43-23eed3a28b8d","resolution":{"observed_at":"2026-08-07T11:39:49.596584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-06T18:01:31.723021Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T11:39:51.726608Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:51.726608Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:fa44a30ec18bbb2f196a6adef49bc96e0801f752d17294c13856c0e5564b3a16","observation_id":"fbe9a7a8-7549-4797-90cc-c253e3ea3f93","resolution":{"observed_at":"2026-08-07T11:39:51.726608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.565367Z","title":"Introducing claude, 2023","venue":null,"work_id":"eefc4a14-f3e8-4c74-b16a-9ee67fde6943","year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:51.844902Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:502df44b9094474f55e2a5106aab6d0758c991e78c5aece8f6bf4ef594d17bfd","observation_id":"172ecc52-7a1f-4e7e-9dd4-2647792a79d0","resolution":{"observed_at":"2026-08-07T11:40:59.637858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.417577Z","title":"Ensemble of averages: Improving model selection and boosting performance in domain generalization","venue":null,"work_id":"12c92548-f2ef-4e2a-9f9b-b56ae3caf6d9","year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:51.935450Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:4c2c2c9dd69bc49d3d723cefde394d172552c5338e83ffd95cd32fee62734679","observation_id":"3ffdb803-0a43-4efb-a25a-1ff12d9cda26","resolution":{"observed_at":"2026-08-07T11:40:59.456395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.11300","last_updated":"2020-01-29T21:01:57Z","snapshot_observed_at":"2026-08-09T11:24:03.051310Z","submitted_at":"2019-06-26T19:09:56Z","title":"Benign Overfitting in Linear Regression","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.11300","snapshot_observed_at":"2026-08-07T11:39:52.065614Z","title":"L., Long, P","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.065614Z"},"links":{"cited_paper":"/paper/1906.11300","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:89fed07cf378e48a26f41829016ea47b28af967f950f949594500ed412327662","observation_id":"afbcd0f9-fedc-4559-97d9-f275908eee4b","resolution":{"observed_at":"2026-08-07T11:39:52.065614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.307099Z","title":"L., Tino, P., and Bengio, Y","venue":null,"work_id":"2c42315c-2a34-499d-9a47-82a50e239a65","year":2005},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.147284Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:452974d88a25fa576be8fbab22c810d82366d46c417463a96c36771ae5d62b51","observation_id":"a623298b-e41b-4867-94af-3d03dbf4920d","resolution":{"observed_at":"2026-08-07T11:40:59.370712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.210353Z","title":"Swad: Domain generalization by seeking flat minima","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.210353Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:e34715a233e3bdd9e0900cb80b20b6ff7926d38887a7746267e9afffadf99a22","observation_id":"b5f4ba60-0b12-4aba-be08-e48396d088dd","resolution":{"observed_at":"2026-08-07T11:39:52.210353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.223051Z","title":"Dna: Domain generalization with diversified neural averaging","venue":null,"work_id":"e9c80831-d89e-4619-ad04-982947eb6c0f","year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.294098Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:6a4512f8472ead41ed9bbd1c7f5593a06ba094ac7c4f58cd80ba4ab553426c0c","observation_id":"d75cd340-413b-40c3-97d3-dddc2076a2ea","resolution":{"observed_at":"2026-08-07T11:40:59.244896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.360435Z","title":"Free dolly: Introducing the world's first truly open instruction-tuned llm, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.360435Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:f8d7e732ae63e38086f0f02894d2174d9cfed884a5a0eae85ac8e730872a9bfb","observation_id":"0dd1e839-fe2d-43fe-b00e-24902ce1aecb","resolution":{"observed_at":"2026-08-07T11:39:52.360435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.185934Z","title":null,"venue":null,"work_id":"1023ab86-dd1d-421f-8a04-8cef8167eb01","year":2000},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.426938Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:4d97179775b62883710c3f53ba838684aca2aa722939571f02cc05beba2fef9e","observation_id":"df60d467-e687-4776-965e-b14abdd4a70a","resolution":{"observed_at":"2026-08-07T11:40:59.210129Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:39:52.493455Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.493455Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:fa901ffdc71f9687b994148d19bda1f341247f3dd7b9af32587d7e2fa00f9149","observation_id":"d0447fd7-df63-436b-a19a-b53ca98a6d97","resolution":{"observed_at":"2026-08-07T11:39:52.493455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:59.085104Z","title":"and Wang, Z","venue":null,"work_id":"8bb17d7e-71a4-4408-9e11-16c20d1a65b4","year":2020},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.672598Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:e17c3d8f47b4e01707cf6883cb30288cf9e365416dc5076a5c7657d42d9ef342","observation_id":"30a62d9e-956f-4879-980d-7e72607afe2f","resolution":{"observed_at":"2026-08-07T11:40:59.124927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6211","last_updated":"2015-03-04T01:43:31Z","snapshot_observed_at":"2026-07-06T03:31:33.797310Z","submitted_at":"2013-12-21T06:31:41Z","title":"An Empirical Investigation of Catastrophic Forgetting in Gradient-Based Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.6211","snapshot_observed_at":"2026-08-07T11:39:52.849538Z","title":"J., Mirza, M., Xiao, D., Courville, A., and Bengio, Y","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.849538Z"},"links":{"cited_paper":"/paper/1312.6211","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:0cd2590031c851ef489b2ba2d935437f993725a6a921a8ac7c42d1266f20b70d","observation_id":"5edb5240-7d46-4122-8e92-94c8d9bcc813","resolution":{"observed_at":"2026-08-07T11:39:52.849538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.903474Z","title":null,"venue":null,"work_id":"7c1bf404-5227-43d4-bc4a-ed56b2734dea","year":1990},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:52.977001Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:6c99a301888e4108b294a344bc1e33c2adca01a12afeecc60f2da015fe446c29","observation_id":"13399c44-75fc-4202-ae74-43f23d80c91a","resolution":{"observed_at":"2026-08-07T11:40:58.992797Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17592","last_updated":"2024-03-26T11:01:53Z","snapshot_observed_at":"2026-08-08T05:13:09.838187Z","submitted_at":"2024-03-26T11:01:53Z","title":"On the Benefits of Over-parameterization for Out-of-Distribution Generalization","version":1},"cited_work":{"arxiv_id":"2403.17592","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.17592","snapshot_observed_at":"2026-08-07T11:40:53.248362Z","title":"On the Benefits of Over-parameterization for Out-of-Distribution Generalization","venue":"cs.LG","work_id":"6de8fe04-c359-40ee-ad62-2051424361fc","year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:53.052775Z"},"links":{"cited_paper":"/paper/2403.17592","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:11aea2c368647f61cf5b744eb995d2951b4654ca2cf675751dfaf2c76173340a","observation_id":"aa270992-d5f1-4f8a-b5ba-199188ccccad","resolution":{"observed_at":"2026-08-07T11:40:54.881523Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-09T10:28:06.906299Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T11:39:53.162961Z","title":"Measuring massive multitask language understanding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:53.162961Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:26fa987df9777c280a36f2f11200c9efa644672f0f4daa2ff7cd006ce245ffbe","observation_id":"59ccb0a6-a42e-4b8e-b19f-41706396d969","resolution":{"observed_at":"2026-08-07T11:39:53.162961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-07T11:39:53.275424Z","title":"J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:53.275424Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:1fae370527eac64d7f2522fbdbe0526a4b33042ae4db80fed1be0596a0cf8585","observation_id":"e60fce52-5f88-4bb3-9d2e-6b8c55e7eb7f","resolution":{"observed_at":"2026-08-07T11:39:53.275424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05489","last_updated":"2021-06-11T00:33:19Z","snapshot_observed_at":"2026-08-04T11:43:57.909017Z","submitted_at":"2021-04-12T14:17:43Z","title":"Continual Learning for Text Classification with Information Disentanglement Based Regularization","version":2},"cited_work":{"arxiv_id":"2104.05489","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.05489","snapshot_observed_at":"2026-08-07T11:40:36.135745Z","title":"Continual Learning for Text Classification with Information Disentanglement Based Regularization","venue":"cs.CL","work_id":"c35be952-dd40-4c04-a518-34bf131a61df","year":2021},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:53.445255Z"},"links":{"cited_paper":"/paper/2104.05489","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:11ec68978004f2dc7900bce4fbe35548b312b27427d11b35c601265b059a3767","observation_id":"461e2c1f-572d-4de4-a57f-cdba94f3caa3","resolution":{"observed_at":"2026-08-07T11:40:46.567716Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.728110Z","title":"and Lounici, K","venue":null,"work_id":"b271726b-95f1-4c64-917e-52ad52530465","year":2017},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:08.018396Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:2cb127edad4d6d2a8bf867977e054092b2b485785a5b36d0774ab06750112cd0","observation_id":"fd6814d4-2475-4181-aa9c-330712df7687","resolution":{"observed_at":"2026-08-07T11:40:58.810355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.653487Z","title":"and Vedelsby, J","venue":null,"work_id":"6cbb4c4b-afda-426b-b876-64d03871fb34","year":1994},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:08.043293Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:83d25e46d94d2e882be22db5ffe25a6e9347731cfcf78015eede64691dd09f3e","observation_id":"3dbb4679-0115-4db8-8568-2985ac3c12c0","resolution":{"observed_at":"2026-08-07T11:40:58.658689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.511228Z","title":"Calibrated ensembles can mitigate accuracy tradeoffs under distribution shift","venue":null,"work_id":"9d13b19f-b386-49db-8089-e8007200d0f5","year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.312558Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:6fd67699d802ff79ed8a4d9eb146281c3a375daaae1d474cb030367d0a372bf4","observation_id":"dfe6c01c-6e39-4022-af6d-5cc5de58d595","resolution":{"observed_at":"2026-08-07T11:40:58.621944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.10054","last_updated":"2022-02-21T09:03:34Z","snapshot_observed_at":"2026-08-09T11:23:03.445810Z","submitted_at":"2022-02-21T09:03:34Z","title":"Fine-Tuning can Distort Pretrained Features and Underperform Out-of-Distribution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.10054","snapshot_observed_at":"2026-08-07T11:40:09.642210Z","title":"Fine-tuning can distort pretrained features and underperform out-of-distribution","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.642210Z"},"links":{"cited_paper":"/paper/2202.10054","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:29ec8769e8c5d7c190e83551a99921fddaa8d547e14b8b454373fd728d94c7d7","observation_id":"502ae3d2-093e-4b5f-806f-b463b3d8b825","resolution":{"observed_at":"2026-08-07T11:40:09.642210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.358822Z","title":"On the eigenvalue decay rates of a class of neural-network related kernel functions defined on general domains","venue":null,"work_id":"b6dafd11-a201-4b3e-b965-9286692698e8","year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.777555Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:dea15f422956dc083a778827b667dd39fda301b9129f34819bedd4947116d2b4","observation_id":"0a1b59d3-630d-4568-a885-1b1564e0e05c","resolution":{"observed_at":"2026-08-07T11:40:58.439268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.219975Z","title":"and Rosasco, L","venue":null,"work_id":"30d6fab5-147d-4609-b46d-d9fa701e9a23","year":2017},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.867929Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:24384635fedaa37bcc210c5d0cbcde4234c3b7a02b24df1f4579bd4de0152601","observation_id":"26fc5a0f-46c1-4233-ac63-662141cbe0ae","resolution":{"observed_at":"2026-08-07T11:40:58.288402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17230","last_updated":"2024-07-14T08:02:49Z","snapshot_observed_at":"2026-07-06T16:25:29.858870Z","submitted_at":"2023-09-29T13:29:22Z","title":"Spurious Feature Diversification Improves Out-of-distribution Generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17230","snapshot_observed_at":"2026-08-07T11:40:09.899709Z","title":"Spurious feature diversification improves out-of-distribution generalization","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.899709Z"},"links":{"cited_paper":"/paper/2309.17230","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:7f046b91cc76a679f4d7a60c96644f972c51f323a80f764c64b13b6335029639","observation_id":"96fbe0d2-f1e5-4ee2-a6ba-f0f65aecc7c0","resolution":{"observed_at":"2026-08-07T11:40:09.899709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06256","last_updated":"2024-10-13T19:27:20Z","snapshot_observed_at":"2026-07-06T16:17:26.924098Z","submitted_at":"2023-09-12T14:16:54Z","title":"Mitigating the Alignment Tax of RLHF","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06256","snapshot_observed_at":"2026-08-07T11:40:09.918152Z","title":"Mitigating the alignment tax of RLHF","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.918152Z"},"links":{"cited_paper":"/paper/2309.06256","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:f4313c0b951e1943f8c0ee91331c1e48f51be813def01d34e18a38c463d16557","observation_id":"68814b27-f0a7-4b71-bc79-75bc0a926d01","resolution":{"observed_at":"2026-08-07T11:40:09.918152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.096711Z","title":"Sobolev acceleration and statistical optimality for learning elliptic equations via gradient descent","venue":null,"work_id":"523e338c-4628-403e-80fb-29cad79e78e4","year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:09.979681Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:d5a101880582c8515b295a963d469132e62ad61a83aea8dfafa7b80960b00f15","observation_id":"92e0c967-f4c2-4c06-823b-7366fe92e71b","resolution":{"observed_at":"2026-08-07T11:40:58.149471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.06569","last_updated":"2024-07-15T21:54:26Z","snapshot_observed_at":"2026-08-04T11:56:00.997924Z","submitted_at":"2022-07-14T00:23:01Z","title":"Benign, Tempered, or Catastrophic: A Taxonomy of Overfitting","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.06569","snapshot_observed_at":"2026-08-07T11:40:11.028961Z","title":"B., Abedsoltan, A., Pandit, P., Belkin, M., and Nakkiran, P","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.028961Z"},"links":{"cited_paper":"/paper/2207.06569","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:c3ba461db496cbcb1b4a185c69382c60d203a786ee0296567a9b551c26f7ae61","observation_id":"5024b512-d4cb-4687-bad6-026f686bf2d2","resolution":{"observed_at":"2026-08-07T11:40:11.028961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:11.633850Z","title":"and Cohen, N","venue":null,"work_id":null,"year":1989},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.633850Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:1be1874beef29d65fcd7d66f976eed217a7fe168a32b5ad915f5c7beab07bcbe","observation_id":"29401c97-275b-4a33-a290-4b6ed00bfcb1","resolution":{"observed_at":"2026-08-07T11:40:11.633850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.056568Z","title":"and Maclin, R","venue":null,"work_id":"28f5032a-62cc-4d6a-b557-666fb21a1a17","year":1999},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.768082Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:11bbd57ce94fc36d67fa9be886148707bf4046388968ad0c2b8c7f39a372f45c","observation_id":"412d7fea-b506-4ea5-99d7-5751bb284095","resolution":{"observed_at":"2026-08-07T11:40:58.072358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.910966Z","title":null,"venue":null,"work_id":"e25fc7e3-43bc-4524-9560-f5fa64bda559","year":1995},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.847496Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:1e1d0f92007568b61cf036a043b7444ebd18b7cee3cb0e01abb6fd40762af253","observation_id":"50c7f944-9560-4fba-ac91-1fd60d7a188c","resolution":{"observed_at":"2026-08-07T11:40:57.962220Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.832590Z","title":"Ensemble based systems in decision making","venue":null,"work_id":"3d850670-945b-4a85-ae2e-c53c7dffd2dc","year":2006},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.909571Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:b298d2d49b5ac7afb88151d1ce9311e98a6ad16899a193d4665ed65c5dc897a3","observation_id":"f56a1cf0-6d8e-41df-ab47-dc3f06c2788f","resolution":{"observed_at":"2026-08-07T11:40:57.895267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.09739","last_updated":"2023-01-27T14:21:58Z","snapshot_observed_at":"2026-08-03T04:00:21.239050Z","submitted_at":"2022-05-19T17:44:22Z","title":"Diverse Weight Averaging for Out-of-Distribution Generalization","version":2},"cited_work":{"arxiv_id":"2205.09739","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.09739","snapshot_observed_at":"2026-08-07T11:40:35.986767Z","title":"Diverse Weight Averaging for Out-of-Distribution Generalization","venue":"cs.CV","work_id":"68c59977-aae9-472e-b54d-ca4fb2b28896","year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.969109Z"},"links":{"cited_paper":"/paper/2205.09739","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:0b5077fcbd58cc9843770e5ea96c420d6907071bdef6892f64c77621ae373db0","observation_id":"b82ce9c2-ca8d-40cf-9f8e-0aa80296ea9d","resolution":{"observed_at":"2026-08-07T11:40:36.043502Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.757588Z","title":"Ensemble-based classifiers","venue":null,"work_id":"4f27d964-f82b-43e5-a1cf-f37c09f12d7c","year":2010},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:11.986156Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:3e55303f7e6efe766a383d9c2eae928c02bf8feed8b9a2265d58955f3851bb75","observation_id":"a09aa34c-fbf8-45e0-8957-fc581aa3ef11","resolution":{"observed_at":"2026-08-07T11:40:57.793081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00937","last_updated":"2019-03-15T18:02:58Z","snapshot_observed_at":"2026-08-04T21:50:24.483867Z","submitted_at":"2018-11-02T15:34:29Z","title":"CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00937","snapshot_observed_at":"2026-08-07T11:40:12.004089Z","title":"Commonsenseqa: A question answering challenge targeting commonsense knowledge, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.004089Z"},"links":{"cited_paper":"/paper/1811.00937","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:a85388918a91eb9a290eb037b92f92b5b207c0d46f180bb7931a64df93e0c18d","observation_id":"3403b045-0edf-4a4f-b37d-faa9a8f72054","resolution":{"observed_at":"2026-08-07T11:40:12.004089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T11:40:12.025309Z","title":"M., Hauth, A., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.025309Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:c552d2db0a03f4e5d74a11cab5201d479dc4bc5d553d0ef24e78233df67d7c1b","observation_id":"a1b21912-b17d-40b3-8309-e7fcfe973bea","resolution":{"observed_at":"2026-08-07T11:40:12.025309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T11:40:12.048112Z","title":"G., Hardin, C., Bhupatiraju, S., Hussenot, L., Mesnard, T., Shahriari, B., Ramé, A., Ferret, J., Liu, P., Tafti, P., Friesen, A., Casbon, M., Ramos, S., Kumar, R., Lan, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.048112Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:0fbbf929efd29494fc54732fffe90d85c6f01f30cf5ea08f197cfca4dcd62db4","observation_id":"0983513c-57e7-443e-9e62-e1cb1ceb7f89","resolution":{"observed_at":"2026-08-07T11:40:12.048112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.528583Z","title":"Trainable projected gradient method for robust fine-tuning","venue":null,"work_id":"10f9dade-4cee-4664-869f-d1f6969d0ce3","year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.074551Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:aeab00ebac4be3e7598c2a6b17935796325f09e033144ab388665585f25a2f56","observation_id":"4e8e40e0-749b-43d2-b57e-5cb40be234b7","resolution":{"observed_at":"2026-08-07T11:40:57.702703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.106681Z","title":"High-dimensional probability: An introduction with applications in data science, volume 47","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.106681Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:1f4c808d4d7bcbb5c60845cf3c07e123a42bcc94070160f76b89c61b9b1c95f4","observation_id":"c2fe79d5-bc5b-4a52-996f-a2d3ae08d708","resolution":{"observed_at":"2026-08-07T11:40:12.106681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.140101Z","title":"Y., Roelofs, R., Gontijo-Lopes, R., Morcos, A","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.140101Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:8efee14f2e27742584ca92753770915281ffc58992297460e104274ff4b55d6e","observation_id":"45f53e01-7c9b-4b4e-be90-45db3a2b83bf","resolution":{"observed_at":"2026-08-07T11:40:12.140101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.195608Z","title":"W., Li, M., Kornblith, S., Roelofs, R., Lopes, R","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.195608Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:975058f28ef475690b1b42439446875fd08af4ca03102e0bca591b74c2c27bab","observation_id":"78a9bb0f-4cd7-4078-aaab-1d2b798e2cdd","resolution":{"observed_at":"2026-08-07T11:40:12.195608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-07T11:40:12.252491Z","title":"Qwen2 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:12.252491Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:b672ce3cc4ff6885d8277e3763426fef7a70a61906e4c2e5fd30c3f8eaff7e7f","observation_id":"c3f78c18-00f8-48a5-9eec-3e51126a2e86","resolution":{"observed_at":"2026-08-07T11:40:12.252491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.997295Z","title":null,"venue":null,"work_id":"eb669476-988e-47cd-a861-5d32c0121205","year":2020},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:21.608115Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:799b49c48d43d0db909fbdcacd1c36579d017c455da53185bf54da76a939c2d8","observation_id":"a1f6c11a-2820-4829-bf61-09074179b341","resolution":{"observed_at":"2026-08-07T11:40:56.045956Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:23.219237Z","title":"Mathematical Analysis of Machine Learning Algorithms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:23.219237Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:f04bf9f67e9202101d51e44c2639dca2128d5ce502db0aa6bc08e2256092f633","observation_id":"aff6a23a-acf1-49c9-91ce-4af0752490ef","resolution":{"observed_at":"2026-08-07T11:40:23.219237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.918217Z","title":"Why transformers need adam: A hessian perspective","venue":null,"work_id":"26bfdd3d-c209-4357-9af1-0f947fe4cdbb","year":2024},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:23.559540Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:577242476bfec26dd2501dfee08b74c1620d4befb109cca71cf6a8d7d6c040dc","observation_id":"1ecd89ef-2210-4916-9e41-2b3d9433fdac","resolution":{"observed_at":"2026-08-07T11:40:55.929003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T11:40:23.793987Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:23.793987Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:4a14b16ed7bebe8f8d544b3db5f7923761d374be8ed84b73a08c04924728b2eb","observation_id":"6b8bbcb3-4434-4ac0-a050-371f282b13aa","resolution":{"observed_at":"2026-08-07T11:40:23.793987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-07T11:40:30.825902Z","title":"P., Zhang, H., Gonzalez, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:30.825902Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:867edc97e3990170caf53e6099eb6ed5831dbca7c9ec15298c13e3dc15e130a3","observation_id":"5ec7e910-45c3-484d-9dfd-d0ff6d089213","resolution":{"observed_at":"2026-08-07T11:40:30.825902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.590792Z","title":"Ensembling neural networks: many could be better than all","venue":null,"work_id":"9adddd3d-40e7-429c-b1f3-bf8ffed95f71","year":2002},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:31.874737Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:0d8babe535d7c97f5023779c3490b96c1b0a44acc0a60215e40118cd2f4bbff4","observation_id":"6303b9e6-ec5b-4b1f-aaf7-5d5a839e0d17","resolution":{"observed_at":"2026-08-07T11:40:55.869485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:33.529471Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:33.529471Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:2433edd3010a12e04f25784e2289b6a9982de0891b9e544585ddbf266dcae67c","observation_id":"79daec92-4007-4a8f-8ff5-cbad6c31145a","resolution":{"observed_at":"2026-08-07T11:40:33.529471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.694751Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:35.694751Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:e4db970f901bc410777522d20064b44f75b80165eb1e239fc2844473972ad91c","observation_id":"1153f784-d2eb-40d1-9824-92c2d44b8782","resolution":{"observed_at":"2026-08-07T11:40:35.694751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.806185Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:35.806185Z"},"links":{"citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:135995a9ce29eb3891699c36d00984c646a8383a317d924b51fa42d0dd081f58","observation_id":"ef36e271-1f0d-41ee-a66e-7c6a712da0c2","resolution":{"observed_at":"2026-08-07T11:40:35.806185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":32,"verified_exact":3,"verified_fuzzy":17},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 3 inbound Pith citation observations for arXiv:2506.01901."}