{"as_of":"2026-08-18T19:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8cd4a8b323bef3517afdecf3ef4a228f9c7446504cdaf000853f54da0d396215","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T16:57:13.699749Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.19498/citation-record","integrity":"/paper/2508.19498/integrity","json":"/paper/2508.19498/citation-record.json","paper":"/paper/2508.19498"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.540890Z","title":null,"venue":null,"work_id":"e6783400-5836-40c0-97ae-837331c312d3","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.414413Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:d8a4f75f2545519c68b11d05e89213ca53377dc9c560166cc75d824788ea31f2","observation_id":"3451b111-2fb9-4faa-b01a-6f6435524bda","resolution":{"observed_at":"2026-08-15T16:57:14.545170Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.529478Z","title":"Evolutionary Optimization of Model Merging Recipes,","venue":null,"work_id":"3621294a-e251-4a8d-9653-fd583d47f89c","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.419002Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:90978dc9da5381c3740deda9a461de4c7b3cbc726c82601151a4da59b35dbd6c","observation_id":"374b3cce-531f-4ded-8590-faa97b7b3eff","resolution":{"observed_at":"2026-08-15T16:57:14.533393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.517229Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","venue":null,"work_id":"e6e32fda-1d72-4e47-b1a9-832a5f5a6ce8","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.427151Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:c92ebcab12de4d376bcaf983af99e793ff1d05986bc63ea24196327a04205f0b","observation_id":"22990a1d-8268-4d0c-84c8-e6d3144a1d45","resolution":{"observed_at":"2026-08-15T16:57:14.521767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.505295Z","title":"Ensemble of Averages: Improving Model Selection and Boosting Performance in Domain Generalization","venue":null,"work_id":"796bc40d-f098-4690-9317-ff8008ee7699","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.431894Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:699138a0df90f25ca918c84e413679ed6fabca2cc63cf1cc2e2f13019480db5d","observation_id":"552b6cdf-955e-4952-9f3b-714423dec62e","resolution":{"observed_at":"2026-08-15T16:57:14.508984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.493245Z","title":"On the Inductive Bias of Neural Tangent Kernels","venue":null,"work_id":"71ffcc57-0183-453f-a7ca-cd0a1e12c96b","year":2019},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.436434Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:fdf6170bbf5949fbbdba21cfe170e9786902a2e6fd92dfd27dd0863b7f27014f","observation_id":"aeaafbb5-91e5-4059-b761-5a44b7df352d","resolution":{"observed_at":"2026-08-15T16:57:14.497654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.481816Z","title":"Food-101 – Mining Discriminative Components with Random Forests","venue":null,"work_id":"dc24a172-60b6-4bd0-80a3-105b497165b0","year":2014},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.440061Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:ee23d9bc8c3a27a1e53fa09ec75154c3f39c1750778b66b1d56efab01196ac15","observation_id":"a262fdd1-f14c-4f3b-a728-0709e217a1a9","resolution":{"observed_at":"2026-08-15T16:57:14.486003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.470830Z","title":"SWAD: Domain Generalization by Seeking Flat Minima","venue":null,"work_id":"a0fd5245-a69f-4bc2-8043-27f7587c3274","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.443614Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:df8332802e3d461cdc5dd55f088953d907f79fce3ff1a9dd9be45d03be93aaa4","observation_id":"07d1fd59-f437-4d68-bdfd-d1112c0df77d","resolution":{"observed_at":"2026-08-15T16:57:14.474455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.459942Z","title":"Distilling Knowledge via Knowledge Review","venue":null,"work_id":"c964abd4-5fd2-4897-bff6-196f55694762","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.446991Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:9b64b04241eb04e5e7aa8ec56caac43586633006bb1700c907df6a1989acfcf8","observation_id":"04f83b44-70dd-4f16-95a1-b6540c5beab4","resolution":{"observed_at":"2026-08-15T16:57:14.463758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.03044","last_updated":"2022-04-06T18:54:48Z","snapshot_observed_at":"2026-08-16T17:08:46.638369Z","submitted_at":"2022-04-06T18:54:48Z","title":"Fusing finetuned models for better pretraining","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.03044","snapshot_observed_at":"2026-08-15T16:57:13.451014Z","title":"Fusing finetuned models for better pretraining, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.451014Z"},"links":{"cited_paper":"/paper/2204.03044","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:d77b22dbdb4e48e35f2a8ed16e3eba68292c9bf1b3a4b1d9a77475fe1c143333","observation_id":"e9b5e894-49bd-40f7-9b97-bae2e9db8703","resolution":{"observed_at":"2026-08-15T16:57:13.451014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.449434Z","title":"Cimpoi, S","venue":null,"work_id":"508c9863-4179-43a3-b915-655b3dc10d8d","year":2014},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.455034Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:5a8e6450c4d1a0ede275984381ee7602adbdd7a48d565c1b5cdda05539589e1a","observation_id":"b4c19dbb-879c-4a42-959e-cb7751413c7b","resolution":{"observed_at":"2026-08-15T16:57:14.452867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.439159Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":null,"work_id":"96c68278-ad74-4777-8cfb-eed0337fc4e4","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.466919Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:a9c57c4f136a9e6b47ad318606be693f0cb99aab9aa0f3f3d2def268eede562a","observation_id":"e6fedc35-5aa3-4cb9-a061-604266465e73","resolution":{"observed_at":"2026-08-15T16:57:14.442640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.427562Z","title":"Agree to Disagree: Adap- tive Ensemble Knowledge Distillation in Gradient Space","venue":null,"work_id":"7bcaaab2-4305-4f0e-9037-df3ae7247e24","year":2020},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.470982Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:662ca5f183daff8d41487243009d6050801da031183866113be89175c60dd9d3","observation_id":"7eb0c7f3-ab49-446b-9041-184862981937","resolution":{"observed_at":"2026-08-15T16:57:14.431426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.474435Z","title":"Learning Factored Representations in a Deep Mixture of Ex- perts","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.474435Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:b76c496f1d6a61bb7e6ce5658119dc26a857d90660d07c7a8f4bf7d98d17de95","observation_id":"c975dcae-7912-4503-9ff0-94a3ff023f80","resolution":{"observed_at":"2026-08-15T16:57:13.474435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05665","last_updated":"2023-05-31T04:57:12Z","snapshot_observed_at":"2026-08-18T09:52:42.083004Z","submitted_at":"2023-05-09T17:59:07Z","title":"ImageBind: One Embedding Space To Bind Them All","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05665","snapshot_observed_at":"2026-08-15T16:57:13.478136Z","title":"ImageBind: One Embedding Space To Bind Them All, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.478136Z"},"links":{"cited_paper":"/paper/2305.05665","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:856b3077912f8f90aa543edbad7d596bc7670df70aafe0c93ca30920c7c5bcc7","observation_id":"3498c0d6-7ae2-421d-9c91-e49c7bec534f","resolution":{"observed_at":"2026-08-15T16:57:13.478136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.409396Z","title":"Borgwardt, Malte J","venue":null,"work_id":"cb335cf1-cd28-4a20-987f-59b0614ed4ba","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.482123Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:91e8fb177caa25824dcc6b11525367b316e807ddb63ef33e05df15e1d02f6ab8","observation_id":"9c0f8f2a-b622-41b8-8f47-05cf21471666","resolution":{"observed_at":"2026-08-15T16:57:14.413445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.398185Z","title":"STOCHASTIC WEIGHT A VERAGING IN PARAL- LEL: LARGE-BATCH TRAINING THAT GENERALIZES WELL","venue":null,"work_id":"e3401a73-2065-4051-89b7-b294e3426c2c","year":2020},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.486134Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:3ef4f927d197d4f89be963e3b7002da7d0311f8bbe088b58e4de9e2281be23b7","observation_id":"e70478d2-d426-4005-a353-a2f93b42a794","resolution":{"observed_at":"2026-08-15T16:57:14.402366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.387078Z","title":"Learning Efficient Vision Transformers via Fine-Grained Manifold Distillation","venue":null,"work_id":"cc43877e-40ac-4752-a24e-d98f94194ccf","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.489971Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:8e61e5f98f4433d227266440d01d6f945d7b1d74618ff7c79452ff97374367cd","observation_id":"65f209f4-0e32-4ac0-821e-cd93f42c2445","resolution":{"observed_at":"2026-08-15T16:57:14.391314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.375563Z","title":"One-for-All: Bridge the Gap Between Heterogeneous Architectures in Knowledge Distil- lation","venue":null,"work_id":"d890cb2a-2c17-42af-b906-28c2fb6dac22","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.493704Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:d79cf0ce60acff68f6dda611cf30938cac73a2a67767c3ed7c6a36f3ab8abcbd","observation_id":"80cd3c37-176f-4be0-a439-07df42f9f7c7","resolution":{"observed_at":"2026-08-15T16:57:14.379706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.364153Z","title":"Deep Residual Learning for Image Recognition","venue":null,"work_id":"5598a3e1-d068-4b77-bcc8-9d3edee387e4","year":2016},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.497161Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:37b5f5d95ec24985479a049d9dce6e795f62092e118dc0721fc570aa0b352387","observation_id":"0585f261-cda0-4a60-99c3-1e4f73ce2838","resolution":{"observed_at":"2026-08-15T16:57:14.367987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.353079Z","title":"A Comprehensive Overhaul of Feature Distillation","venue":null,"work_id":"ef743572-6e16-4ee6-8057-e44585398b2b","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.500703Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:70f3b10ce8e5ffbc851aeb422d87d99076172fdcff0e6b9d05a3d7227e9e60b6","observation_id":"15efad14-5f55-446e-b3e0-def278f0a5e7","resolution":{"observed_at":"2026-08-15T16:57:14.356970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-08-16T18:00:58.008096Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-15T16:57:13.504099Z","title":"Distilling the Knowledge in a Neural Network, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.504099Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:e5f40d557d6edb2d436345f352b18a9ea5a603094615ad9d221fbceead62ab44","observation_id":"a8bcf93e-6e19-49ba-81a6-3c2c9537975c","resolution":{"observed_at":"2026-08-15T16:57:13.504099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.342253Z","title":"Edit- ing models with task arithmetic","venue":null,"work_id":"1b2df50e-7c61-472a-886d-1f1ae08f1d82","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.507951Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:e3dde68be349bbf4dc0aee88b07fdf43d7b2ca30ac5e40254c4dcb5ba7a40cf0","observation_id":"cb81d2d4-780a-4d15-b36e-a34738902dea","resolution":{"observed_at":"2026-08-15T16:57:14.346241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.331908Z","title":"Averaging Weights Leads to Wider Optima and Better Generalization","venue":null,"work_id":"0c431491-07d2-466b-9ceb-647f3829a3b5","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.511247Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:66f6584adf77a09a3e8c4a5bcaca8d280c0952df92167e7ae4d02fc32d4623c3","observation_id":"6fa57c50-cb1e-4772-91ec-5fa3d7690be7","resolution":{"observed_at":"2026-08-15T16:57:14.335586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.321725Z","title":"Dataless Knowledge Fusion by Merging Weights of Language Models","venue":null,"work_id":"ad07ff64-f8ec-49ab-a826-adc53001bd6b","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.515419Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:58930643d1323a70d7f708fc6aaf8b69279c595cfacb1895a8fc3900c6e9ae34","observation_id":"ab13bca7-538b-487e-8302-19d4d8fdd468","resolution":{"observed_at":"2026-08-15T16:57:14.325230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.311199Z","title":"Khosla, N","venue":null,"work_id":"6c34dd13-81a6-4442-8cce-fe3c38b65a0f","year":2011},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.519240Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:af930407c2ba5e73a29fe112df1a4956abc5cbc3bc1bfe08c14c324bf9ef6ab3","observation_id":"150ccff6-22e3-41c3-98ce-b1c2a77ff753","resolution":{"observed_at":"2026-08-15T16:57:14.314875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.300461Z","title":"3D object representations for fine-grained categorization","venue":null,"work_id":"68faf036-567e-4441-8d91-7f260bbb3064","year":2013},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.522926Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:085a1542c72062bba6c349800006115bb477f686cc61a2c01d7d9b0775fe05eb","observation_id":"159820da-ea04-4da5-8052-a81fcce2dcd8","resolution":{"observed_at":"2026-08-15T16:57:14.304143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.289611Z","title":"Learning multiple layers of features from tiny images","venue":null,"work_id":"64c16aa7-95dd-45d8-be2e-c852cdbe1c18","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.526630Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:2be5e5049c895255f26f521159b990b1421425270e382036189eed351fb34464","observation_id":"eab18b72-ba68-487d-9a72-fd1a846307df","resolution":{"observed_at":"2026-08-15T16:57:14.293338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.279128Z","title":"Caltech 101, 2022","venue":null,"work_id":"0ae94913-e1cc-4b86-9c7f-0dea6b0b6f7e","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.530373Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:6f5b4bfc98cbc1deb7d8fb3942405535e12f6eeea470db4c95aea341d5386ee6","observation_id":"038a901c-73d3-4b6c-946b-c5cb3011f9ad","resolution":{"observed_at":"2026-08-15T16:57:14.282462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.03306","last_updated":"2022-08-05T17:46:38Z","snapshot_observed_at":"2026-08-18T13:05:31.068739Z","submitted_at":"2022-08-05T17:46:38Z","title":"Branch-Train-Merge: Embarrassingly Parallel Training of Expert Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.03306","snapshot_observed_at":"2026-08-15T16:57:13.533986Z","title":"Smith, and Luke Zettlemoyer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.533986Z"},"links":{"cited_paper":"/paper/2208.03306","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:6a675171ea0ed10ad4f111af0ae9b7fffc910b146bf0d6b7b7e2ac51906735b9","observation_id":"1fb39a41-4564-4b72-860a-0e6321a981fe","resolution":{"observed_at":"2026-08-15T16:57:13.533986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.268998Z","title":"Merge, Then Compress: Demystify Efficient SMoE with Hints from Its Routing Policy","venue":null,"work_id":"0d761bc4-f863-45cd-9eee-83157c4ea6db","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.537917Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:523dae0f810adc1c5c69da4ff98922725754e099867dea6ed3d5505e1f34ed72","observation_id":"22aab37a-e6e2-4881-ab17-b8728d89b48c","resolution":{"observed_at":"2026-08-15T16:57:14.272536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.258437Z","title":"Harmonious atten- tion network for person re-identification","venue":null,"work_id":"fb92ae09-bcf2-4dad-af8c-dc5a55d2a25e","year":2018},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.541575Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:df3811df188c47e56ebc70c8efb4f9ec6b5998fd238ef2bad7991b5fcc929eb2","observation_id":"075fcbf8-359b-4318-8e34-851820007d69","resolution":{"observed_at":"2026-08-15T16:57:14.262188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.247850Z","title":"Implicit Bias of Gradient Descent based Adversarial Training on Separable Data","venue":null,"work_id":"4a0d192e-341e-40a1-ae08-8198e7141207","year":2019},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.545722Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:b1ee7207287763fd64e47282970814576341fe0e6fb633dc2310b4f055ecb1ef","observation_id":"87c292a6-19ca-4aa7-8381-1d2697b55e24","resolution":{"observed_at":"2026-08-15T16:57:14.251490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15947","snapshot_observed_at":"2026-08-15T16:57:13.549521Z","title":"MoE-LLaV A: Mixture of Experts for Large Vision-Language Models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.549521Z"},"links":{"cited_paper":"/paper/2401.15947","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:9b9da721812022e3b1f1649eff8f053001b1dc3d66bea0c73beb58df6725951f","observation_id":"ec427b6f-439e-44ab-8958-f865354b75c2","resolution":{"observed_at":"2026-08-15T16:57:13.549521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.235519Z","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","venue":null,"work_id":"885aaf79-1df9-4396-bfb8-f3545dbc5b9b","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.553575Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:f7fcb5420f68fb7e8002a1747a69ff51bad73cb1f94985df75a37cb31fcc6f56","observation_id":"9b2ea5eb-b683-4562-99a6-520da358740c","resolution":{"observed_at":"2026-08-15T16:57:14.240625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.222618Z","title":"A ConvNet for the 2020s","venue":null,"work_id":"4d749917-b1e2-469a-8657-d3b86cbc8c27","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.557049Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:4e5f3f7acd9b9c61d3aa308fcd5d137adf29aba835e96aaa9c5b3f2b50ac3ede","observation_id":"f0277ad6-aa6f-4c1c-b7db-103c2e27aa65","resolution":{"observed_at":"2026-08-15T16:57:14.227911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.211511Z","title":"Knowledge Amalgamation from Het- erogeneous Networks by Common Feature Learning","venue":null,"work_id":"5ac92675-2d06-4306-977e-b1c9259244e8","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.560955Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:55ee98fa6cf19047a36d47a8a2794b2df04e01f0ce40b579f2d3fc758e99bf23","observation_id":"77470bd8-1791-4c1d-a26a-0e2d803dc244","resolution":{"observed_at":"2026-08-15T16:57:14.215557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1306.5151","last_updated":"2013-06-21T14:31:57Z","snapshot_observed_at":"2026-08-12T17:35:23.022229Z","submitted_at":"2013-06-21T14:31:57Z","title":"Fine-Grained Visual Classification of Aircraft","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1306.5151","snapshot_observed_at":"2026-08-15T16:57:13.569169Z","title":"Fine-Grained Visual Classifi- cation of Aircraft","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.569169Z"},"links":{"cited_paper":"/paper/1306.5151","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:94fefee55ec01fccc0fdc71b21013ddb968fffca6f259ad2f9805e818e3e29e2","observation_id":"9cccb4d0-1911-4cbb-94f9-43fb27804193","resolution":{"observed_at":"2026-08-15T16:57:13.569169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.189726Z","title":"Im- proved Knowledge Distillation via Teacher Assistant","venue":null,"work_id":"1f311f94-48b1-43c1-906f-b0c17b496c94","year":2020},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.573166Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:5ca9b819ec87266ca0b309e1d3dabe8d8dffadb88144f633d86e9b00229a4c7c","observation_id":"9ec8bb9f-9edc-48d2-85c5-28f7f1a1a539","resolution":{"observed_at":"2026-08-15T16:57:14.193260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.200835Z","title":"5, 6, 7, 1","venue":null,"work_id":"94344785-022c-4e5a-bd72-54cc778bcf2c","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.564901Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:c51646cbcdaab206336a85b2546940f6f4b13a72d8b339403e1e408bbde3e69a","observation_id":"7fec5701-cd9d-4252-8a1c-c1e0cc66e4dc","resolution":{"observed_at":"2026-08-15T16:57:14.204327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.168747Z","title":"Parkhi, Andrea Vedaldi, Andrew Zisserman, and C","venue":null,"work_id":"cadbaa15-dd9b-4751-a367-ef4c1743bb81","year":2012},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.580544Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:e9d77f84c5291b3c0365d8e041e541899297a898d05aa1a1bbf8016291851414","observation_id":"b9cecff5-f734-4446-b676-86d33afd3c56","resolution":{"observed_at":"2026-08-15T16:57:14.172194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.01802","last_updated":"2019-04-03T06:58:10Z","snapshot_observed_at":"2026-08-14T16:52:34.854286Z","submitted_at":"2019-04-03T06:58:10Z","title":"Correlation Congruence for Knowledge Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.01802","snapshot_observed_at":"2026-08-15T16:57:13.584055Z","title":"Correlation Congruence for Knowledge Distillation, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.584055Z"},"links":{"cited_paper":"/paper/1904.01802","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:6ffe4dad26591ddd3c0c2f6a33caf957f4baa60969bdf32cdc9dbf2a419716e0","observation_id":"d701ad46-10b6-4169-8125-13d2aa3efd4e","resolution":{"observed_at":"2026-08-15T16:57:13.584055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.178971Z","title":"Automated Flower Classification over a Large Number of Classes","venue":null,"work_id":"2f0ebd08-a149-4423-a0e8-518cd30a3b10","year":2008},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.576972Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:482ef0bec99a0d91a29e80d4b393379978efc9681161fb6796ff0aea7dc35319","observation_id":"435c5216-223a-4ad8-b8fe-a17970c746fe","resolution":{"observed_at":"2026-08-15T16:57:14.182459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.145329Z","title":"Diverse Weight Averaging for Out-of-Distribution General- ization","venue":null,"work_id":"3a1d54c3-2350-4a18-81ae-c46d579d8616","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.591699Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:31af31be5c2bd9bbe90827e513d436f9fd0df8aea4c0283e821eb7866097aca3","observation_id":"78038f2b-c989-4288-95ed-19a5c4904212","resolution":{"observed_at":"2026-08-15T16:57:14.149172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.132883Z","title":"Model Ratatouille: Recy- cling Diverse Models for Out-of-Distribution Generalization","venue":null,"work_id":"9a59c5c0-5960-48be-8c6f-fad9d6b10e1c","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.595292Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:f9d2acebd9f9b11c4d14bfe87f8312966bc9725d34a82c2766b95f05f74d7478","observation_id":"68c40faf-0673-43d0-aaee-c958a38bc634","resolution":{"observed_at":"2026-08-15T16:57:14.137160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.157075Z","title":"Learning Transferable Visual Models From Natural Language Supervision","venue":null,"work_id":"c2b49ce8-b00a-457c-aa3f-6f1044768668","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.587937Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:c2aaaf3d8863feb4421555e7b5df27353ee5d45c9dd7445f23c4fd426f42c601","observation_id":"8516403b-d655-4be6-9358-4a8efcfa0e42","resolution":{"observed_at":"2026-08-15T16:57:14.161124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6550","last_updated":"2015-03-27T11:52:28Z","snapshot_observed_at":"2026-08-17T23:25:45.947983Z","submitted_at":"2014-12-19T22:40:51Z","title":"FitNets: Hints for Thin Deep Nets","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6550","snapshot_observed_at":"2026-08-15T16:57:13.602610Z","title":"FitNets: Hints for Thin Deep Nets, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.602610Z"},"links":{"cited_paper":"/paper/1412.6550","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:08aa62ce789ef67453ff679aa121b743b37de6ce10e3eedf1b95ab765e609944","observation_id":"090944a5-3e84-43b4-abfa-a987ae14835b","resolution":{"observed_at":"2026-08-15T16:57:13.602610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05088","last_updated":"2024-08-09T14:18:57Z","snapshot_observed_at":"2026-08-16T13:27:26.736948Z","submitted_at":"2024-08-09T14:18:57Z","title":"UNIC: Universal Classification Models via Multi-teacher Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05088","snapshot_observed_at":"2026-08-15T16:57:13.606385Z","title":"Unic: Universal classi- fication models via multi-teacher distillation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.606385Z"},"links":{"cited_paper":"/paper/2408.05088","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:0fe3aae28e9c55f89693398c3fe56db2e7052363862f4aee6e7e177fce5f876c","observation_id":"dd6cd192-a0d9-4d6b-96d8-86c423856e5f","resolution":{"observed_at":"2026-08-15T16:57:13.606385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.121582Z","title":"Am-radio: Agglomerative vision foundation model reduce all domains into one","venue":null,"work_id":"1ce9aa79-91c2-4247-8587-dc64885e29bd","year":2024},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.599131Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:abec062d93ef7111bcb2d9c8bd01bfd0399ad644c46b73cf99e77c44f816ba27","observation_id":"8da66ed6-d8da-44bc-b152-f4fe4ed7e000","resolution":{"observed_at":"2026-08-15T16:57:14.125254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.109457Z","title":"Scaling Vision-Language Models with Sparse Mixture of Experts","venue":null,"work_id":"8b00faf6-37d7-4bd8-b075-5a59a6cbc2cc","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.614482Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:84de84f7921c027e156164cced2b9c3b024925a2c3d511765814da1b42f3cc45","observation_id":"95a2ffa3-cd3b-48ab-b1cd-ec8ce6175ac2","resolution":{"observed_at":"2026-08-15T16:57:14.113526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1409.1556","last_updated":"2015-04-10T16:25:04Z","snapshot_observed_at":"2026-08-17T19:17:06.411141Z","submitted_at":"2014-09-04T19:48:04Z","title":"Very Deep Convolutional Networks for Large-Scale Image Recognition","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.1556","snapshot_observed_at":"2026-08-15T16:57:13.618371Z","title":"Very deep convolutional networks for large-scale image recognition","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.618371Z"},"links":{"cited_paper":"/paper/1409.1556","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:682c7bd0a1954a31f7d108b757eed72296f60a6ca8af9467bc14e4a86478f3b7","observation_id":"3cc9b924-ce7b-4d8a-b5a7-78b9b74ca1ba","resolution":{"observed_at":"2026-08-15T16:57:13.618371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.12845","last_updated":"2022-06-26T11:12:49Z","snapshot_observed_at":"2026-08-16T16:50:21.266178Z","submitted_at":"2022-06-26T11:12:49Z","title":"RoME: Role-aware Mixture-of-Expert Transformer for Text-to-Video Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.12845","snapshot_observed_at":"2026-08-15T16:57:13.610491Z","title":"RoME: Role-aware Mixture-of-Expert Transformer for Text-to-Video Retrieval, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.610491Z"},"links":{"cited_paper":"/paper/2206.12845","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:b6721064f4c1ac21b56e12c6127478504db3e362f21f052a269231856c4b45bb","observation_id":"5ea8211e-773d-450c-9473-6fab2d7793e6","resolution":{"observed_at":"2026-08-15T16:57:13.610491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.10699","last_updated":"2022-01-24T19:12:34Z","snapshot_observed_at":"2026-08-10T18:58:30.893017Z","submitted_at":"2019-10-23T17:59:18Z","title":"Contrastive Representation Distillation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.10699","snapshot_observed_at":"2026-08-15T16:57:13.629345Z","title":"Contrastive Representation Distillation, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.629345Z"},"links":{"cited_paper":"/paper/1910.10699","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:70b4542f0233e93984fe42479672f4d513ad5ea00c299cebb4816ab63d3069b0","observation_id":"6b244d18-e9fe-450e-b0ca-348faeb6b7f9","resolution":{"observed_at":"2026-08-15T16:57:13.629345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.077529Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"3570413d-69d8-4b3e-9475-00d0d50f650c","year":2017},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.633129Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:acd65020fa6cb7c86fd3fb48ccc22a2dcab259bc4e0ba2af75e0966f738e2d30","observation_id":"fe6d0b93-68a1-4a4b-ad62-ddcdd31593d3","resolution":{"observed_at":"2026-08-15T16:57:14.081339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.097085Z","title":"An Empirical Study of Multimodal Model Merging","venue":null,"work_id":"5ccf2c22-73c2-4b59-ac52-51df067b192e","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.622390Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:888f927b3e952e7319f9089c3bde9deedd2902f8dbfadabe36e8b487a3ea9be7","observation_id":"bc02e6f7-6134-4290-bb13-cb4880443f2a","resolution":{"observed_at":"2026-08-15T16:57:14.101045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.054638Z","title":"The Implicit Bias for Adaptive Optimization Algorithms on Ho- mogeneous Neural Networks","venue":null,"work_id":"29625ee0-1da9-4422-b8f1-aa42fdb5341d","year":2021},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.641071Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:909083aee4bef30d5b29f527e075cd5376d09613b11e12c2c4d5f7cde31b8883","observation_id":"651d385a-b5d9-4260-8c73-701ec9c75e22","resolution":{"observed_at":"2026-08-15T16:57:14.058375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.042487Z","title":"Momentum Doesn’t Change The Implicit Bias","venue":null,"work_id":"fcd58e57-030a-4df9-84aa-9207d3b7cfae","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.644788Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:ae349255cbb98e5f2c6aa1fc8835f63fa5826385f4f8cfbc19aa88f9cd19e3a8","observation_id":"9f30f923-d7af-416f-93bc-155048b8a973","resolution":{"observed_at":"2026-08-15T16:57:14.047356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15308","last_updated":"2024-06-10T19:19:16Z","snapshot_observed_at":"2026-08-16T14:49:35.613227Z","submitted_at":"2023-10-23T19:21:57Z","title":"SAM-CLIP: Merging Vision Foundation Models towards Semantic and Spatial Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15308","snapshot_observed_at":"2026-08-15T16:57:13.648557Z","title":"SAM-CLIP: Merging Vision Foundation Mod- els towards Semantic and Spatial Understanding, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.648557Z"},"links":{"cited_paper":"/paper/2310.15308","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:4b48f1df2a235ccfc01618e595458aaa7bf7244af0c50866bd27741896c20f98","observation_id":"b21c2835-8e94-4227-b198-c8713e2a382b","resolution":{"observed_at":"2026-08-15T16:57:13.648557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.066134Z","title":null,"venue":null,"work_id":"d10e6de9-bcff-4d79-a3fe-1fefa47191d6","year":2011},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.636929Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:bbd8b0aca915c0bd1fd6cc73594f11054ce571d678edba3bb55514dcff671969","observation_id":"cd4d8453-8bcc-4384-9d56-0af2cb1c9762","resolution":{"observed_at":"2026-08-15T16:57:14.069723Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03284","last_updated":"2023-04-06T17:59:57Z","snapshot_observed_at":"2026-08-16T15:42:11.267112Z","submitted_at":"2023-04-06T17:59:57Z","title":"SegGPT: Segmenting Everything In Context","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03284","snapshot_observed_at":"2026-08-15T16:57:13.656719Z","title":"SegGPT: Segmenting Every- thing In Context, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.656719Z"},"links":{"cited_paper":"/paper/2304.03284","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:ac887782a9472ee8456a21e9a4fa5b412d1bf2338a6a2bc897357f5cc88712d3","observation_id":"8e46fc0b-581e-49ea-9e29-a846414c05ea","resolution":{"observed_at":"2026-08-15T16:57:13.656719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.031558Z","title":null,"venue":null,"work_id":"e017f239-95b9-40c9-a416-361a4c4fb0df","year":2020},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.660453Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:38e2b4f76ac6086d23927ba7d4bee26b7cfe45962c9a3a559ac99a3a98a889d6","observation_id":"19dc4cc7-019f-4a0c-b4ae-d15db6940b6f","resolution":{"observed_at":"2026-08-15T16:57:14.035262Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.019523Z","title":"Morcos, Hongseok Namkoong, Ali Farhadi, Yair Carmon, Simon Kornblith, and Ludwig Schmidt","venue":null,"work_id":"6ad52307-4b24-41dd-aedf-5d168092cc55","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.664384Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:7094b3988a163f85b0f7236f8d80f8c0c2ffa8e49b40c2818353d728158b7150","observation_id":"303a3c26-8da9-495f-b638-0cd04a23563a","resolution":{"observed_at":"2026-08-15T16:57:14.023850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03601","last_updated":"2018-05-25T02:36:16Z","snapshot_observed_at":"2026-08-14T19:47:10.814748Z","submitted_at":"2018-02-10T14:35:27Z","title":"Deep Visual Domain Adaptation: A Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.03601","snapshot_observed_at":"2026-08-15T16:57:13.652486Z","title":"Deep Visual Domain Adapta- tion: A Survey","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.652486Z"},"links":{"cited_paper":"/paper/1802.03601","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:29b7913ef0047910d2b13b1e6c29af7ac79987a2fbda67746afb143f168d8e8c","observation_id":"7378a023-daa3-4c4f-b3b1-6fd111207434","resolution":{"observed_at":"2026-08-15T16:57:13.652486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:14.007328Z","title":"Exclusive Supermask Sub- network Training for Continual Learning","venue":null,"work_id":"4cad1728-91fb-4372-9749-19d87b81f687","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.674038Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:a56a5883ac1b1b15922737cda275b3cdbfe77933965ae4bc98655d603a855d15","observation_id":"4dce647e-71fe-4553-b665-b1ffe4eadf8e","resolution":{"observed_at":"2026-08-15T16:57:14.011550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.995411Z","title":"TIES-Merging: Resolving Interference When Merging Models","venue":null,"work_id":"9f37c036-2653-43f2-aa67-ad580a4015ca","year":2023},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.678545Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:25b16eaffe1159c077f573c4b7dfdbaa0389438ec498fcc35c62ac4f78c2a150","observation_id":"e2ee8fb6-5cdc-49a4-9c98-648fedde6bbc","resolution":{"observed_at":"2026-08-15T16:57:13.999461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.983526Z","title":"Wide Residual Networks","venue":null,"work_id":"c8d09e92-d990-4969-aac0-59da44ac29ab","year":2016},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.683174Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:10e3af11f5746d8b2105753c34ef42f2a7a188f36bb2b33bc4eeec6a176faab8","observation_id":"1c2cfbd8-8888-4b30-bb68-5746cf1fa9af","resolution":{"observed_at":"2026-08-15T16:57:13.987148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00968","last_updated":"2024-04-02T19:57:32Z","snapshot_observed_at":"2026-08-16T14:38:19.603245Z","submitted_at":"2023-12-01T23:04:27Z","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00968","snapshot_observed_at":"2026-08-15T16:57:13.668718Z","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.668718Z"},"links":{"cited_paper":"/paper/2312.00968","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:f10d5825fffc77029a2ff2bfed2352f3e95a674ad27f5a711db1d70bc3a27171","observation_id":"9bb880fd-9aa8-4e9c-8958-4149e4a893fa","resolution":{"observed_at":"2026-08-15T16:57:13.668718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.960978Z","title":"Hospedales, and Huchuan Lu","venue":null,"work_id":"7aec23d8-0cf8-44c7-aef9-832b8745cdc9","year":2018},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.691511Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:89e0e3dcbd07670c63d3250edbe310c976ea92f2b0158409c5101297b357f867","observation_id":"32143dc3-2c81-48d7-a940-c0ed4fa73300","resolution":{"observed_at":"2026-08-15T16:57:13.964343Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.949724Z","title":"Decoupled Knowledge Distillation","venue":null,"work_id":"9623b11b-2098-4c0a-9898-0b39d43f7a14","year":2022},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.695489Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:011a4d3546920c768712dfe887f281731a165f30f847721493b095a98cd44941","observation_id":"29ac1bd4-16ad-40c6-92ee-78b1b5c0a2ef","resolution":{"observed_at":"2026-08-15T16:57:13.953725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.971810Z","title":"Task-Oriented Feature Distillation","venue":null,"work_id":"6b84f308-4fd1-4513-be53-cf24c7e4d8c1","year":2020},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.687224Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:aeba51a6c49495d849b10ae95d4b670c8b16ff44ee1f3322d96d6ea543c0d97e","observation_id":"f3b1d21b-2182-46a7-93ac-44270b978c0a","resolution":{"observed_at":"2026-08-15T16:57:13.976215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.936822Z","title":null,"venue":null,"work_id":"409a3e5b-4266-4f6d-9a79-666b1eeb5ec8","year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.699749Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:f28d8bcdcc55380ccd45714df59cd781452c813ae344a6231f3d34b088c3cd28","observation_id":"fb89fa82-6a2f-4db1-bc6c-46925ce51043","resolution":{"observed_at":"2026-08-15T16:57:13.942173Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-08-14T18:16:28.847993Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-15T16:57:13.463127Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.463127Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:f9b5bc31116c58abda2fcaa78126054a36042411bf904399feee9d156f962459","observation_id":"fcff9645-ca66-400e-89a3-3c2540cc855e","resolution":{"observed_at":"2026-08-15T16:57:13.463127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:57:13.625852Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.625852Z"},"links":{"citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:348f1a8508e557d491f60ee90775bf54e98c04d2589dbca3f3f9f4808ce921a0","observation_id":"fd396793-71e4-40df-8928-329bd251c5c9","resolution":{"observed_at":"2026-08-15T16:57:13.625852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13187","last_updated":"2025-01-27T10:19:44Z","snapshot_observed_at":"2026-08-16T14:08:13.299258Z","submitted_at":"2024-03-19T22:56:53Z","title":"Evolutionary Optimization of Model Merging Recipes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13187","snapshot_observed_at":"2026-08-15T16:57:13.422698Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T16:57:13.422698Z"},"links":{"cited_paper":"/paper/2403.13187","citing_paper":"/paper/2508.19498"},"observation_digest":"sha256:9351357b1ef1c2b4e8c4b0dbe287e1e76d47a70ab6d0b7a4f06f3625deeb3c37","observation_id":"23a79dff-f774-4ae5-aa7b-90b59c48fb57","resolution":{"observed_at":"2026-08-15T16:57:13.422698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.19498","last_updated":"2025-08-27T00:56:11Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T23:26:37.386808Z","submitted_at":"2025-08-27T00:56:11Z","title":"UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":49},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 0 inbound Pith citation observations for arXiv:2508.19498."}