{"as_of":"2026-08-19T04:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bcf05750e287810ccac92afd65617a908c9ecd7d76774f71f6b18be49d6e679b","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:33:37.925834Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:16:14.798670Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T21:16:15.596911Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"cited_work":{"arxiv_id":"2505.12815","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.12815","snapshot_observed_at":"2026-08-06T21:16:15.596911Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","venue":"cs.DC","work_id":"876ddef2-bffd-4df3-a0c1-70116cc832d8","year":2025},"citing_paper":{"arxiv_id":"2507.00672","last_updated":"2025-07-01T11:13:56Z","snapshot_observed_at":"2026-08-14T07:42:42.213803Z","submitted_at":"2025-07-01T11:13:56Z","title":"Toward Edge General Intelligence with Multiple-Large Language Model (Multi-LLM): Architecture, Trust, and Orchestration","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T21:16:14.798670Z"},"links":{"cited_paper":"/paper/2505.12815","citing_paper":"/paper/2507.00672"},"observation_digest":"sha256:5f6d250fe3ab7992f5a33b095b64d30a4474982b4ae1b65b92c824e96f3a4992","observation_id":"734e7039-4347-43df-8efd-e8282c1a0b65","resolution":{"observed_at":"2026-08-06T21:16:15.600083Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.12815/citation-record","integrity":"/paper/2505.12815/integrity","json":"/paper/2505.12815/citation-record.json","paper":"/paper/2505.12815"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:39.033149Z","title":"Federated learning for predicting clinical outcomes in patients with covid-19,","venue":null,"work_id":"342bf3eb-f51a-4ad7-8e61-ad37f8aeb7da","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.218736Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:fb6aa74d82f0dc9217ed13d656298dd375a5b36a3cc6d2bb6c68f45459ca2ec5","observation_id":"bb1a5747-7902-45e9-80ad-de028a7885e7","resolution":{"observed_at":"2026-08-15T20:33:39.050808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.924139Z","title":"Federated learning for breast density classification: A real-world implementation,","venue":null,"work_id":"d2e40a0f-b69a-4a62-aac1-fca14bfb201a","year":2020},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.306429Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:17e6046595d0bc9b1d253b8a4a4c7f8cae9d8bf63c5ccf6ae234343d523df2c4","observation_id":"dd50128a-b417-4578-9018-e9c192fbff55","resolution":{"observed_at":"2026-08-15T20:33:38.994311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.893789Z","title":"Federated learning improves site performance in multicenter deep learning without data sharing,","venue":null,"work_id":"e97b43d1-ddb1-4a20-862e-7ddf0eaf5329","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.404384Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:0cd713941f7a6c1cded41c2467d6820a233c89adfe182b14eb41d4315655d433","observation_id":"e39f0284-eff5-4c88-b58f-6aea57ad6014","resolution":{"observed_at":"2026-08-15T20:33:38.897655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.883185Z","title":"Evaluation of federated learning variations for covid-19 diagnosis using chest radiographs from 42 us and european hospitals,","venue":null,"work_id":"97088673-760b-4520-a423-0f632733f5b0","year":2022},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.447262Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:72e9f51b5820df174b760993c36144d15bd562997f650749ef4645c213ee9d69","observation_id":"a3184c00-78f7-44ad-906b-da48b378493e","resolution":{"observed_at":"2026-08-15T20:33:38.887040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.873058Z","title":"Federated learning enables big data for rare cancer boundary detection,","venue":null,"work_id":"ddf0f828-4867-4218-bc30-bdf73a901687","year":2022},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.483965Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:4e8aed690a7de923f695c668be8d7454b6939087afa15dacd83f1a4b31738d24","observation_id":"fb757215-63c1-4712-9233-7f4c4576efde","resolution":{"observed_at":"2026-08-15T20:33:38.876764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.862058Z","title":"Melloddy: Cross- pharma federated learning at unprecedented scale unlocks benefits in qsar without compromising proprietary information,","venue":null,"work_id":"5a0178ca-d8b4-4bd0-a663-0e49d077f7af","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.487863Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:3aae08366b8222b096a1a7959203f6c10ba46b3dbf37b28fd67b78454aa7d356","observation_id":"84ca8b48-2eef-4ba3-8c95-60a3d27d44d8","resolution":{"observed_at":"2026-08-15T20:33:38.866138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.851249Z","title":"Fedpetuning: When federated learning meets the parameter-efficient tuning methods of pre-trained language models,","venue":null,"work_id":"8428b039-42a8-426e-8e11-2da072924e93","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.492798Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:abc5db140fbb11bc9a7784c64b59ff96197d8f1d520692e39f3f81a3fa5f0aa6","observation_id":"ab256b27-bfb3-4467-bfd1-8e7c4b5c4c3b","resolution":{"observed_at":"2026-08-15T20:33:38.855256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.838086Z","title":"Federated fine-tuning of large language models under heterogeneous tasks and client resources,","venue":null,"work_id":"7ed62eb2-e8ac-4d73-a7f4-3d094e8737ed","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.496183Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:2a8f1e04c37199e0ce070d83606a1d5f27ee8a17b8d801bec317d3a7a47584f0","observation_id":"76f961f7-2736-469f-b3cd-7646cae7b848","resolution":{"observed_at":"2026-08-15T20:33:38.841253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.827821Z","title":"Flora: Federated fine-tuning large language models with heterogeneous low-rank adaptations,","venue":null,"work_id":"9c62b7fe-b4e6-47dd-81a2-c15b57270e65","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.500588Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:de055e8f2b1ce4ed8511d8f89f9bac3213d69c97a42c0acb91387b7b26e63a8f","observation_id":"ee300473-4550-4af8-aedc-b52c96c2e64f","resolution":{"observed_at":"2026-08-15T20:33:38.831709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.817144Z","title":"Federated residual low-rank adaptation of large language models,","venue":null,"work_id":"73f7fc1b-bcfd-41d6-9d50-ceb40155ebec","year":2025},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.504074Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:d7fdddf7cfad8fa2930fe5ef2de11dfa6cdada185ce579a8db4a7e1a6802cb1c","observation_id":"5b642d1d-553b-4e79-b836-d6f5f141da2d","resolution":{"observed_at":"2026-08-15T20:33:38.820931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.733632Z","title":"Fedex-lora: Exact aggrega- tion for federated and efficient fine-tuning of large language models,","venue":null,"work_id":"836977f0-0129-488f-a4c8-9b9b0d7caabb","year":2025},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.507688Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:10a82d76e3aa72e8c8287c1126b980585e324148fa65ae6a6df90154910d365e","observation_id":"ad18427e-56be-4c67-a3a4-e81feef6c52c","resolution":{"observed_at":"2026-08-15T20:33:38.783241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.579717Z","title":"Improving lora in privacy-preserving federated learning,","venue":null,"work_id":"9c4f156b-ee5a-47ba-82a4-8eecdcb1e792","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.512070Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:b5d055863741e69d8ea252fde43f9296b64bbe2cdef386499e5071e2c5b96b01","observation_id":"33ad88c6-0a08-4e5c-82c8-f2ce2139ba50","resolution":{"observed_at":"2026-08-15T20:33:38.622990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.00134","last_updated":"2023-12-30T04:06:16Z","snapshot_observed_at":"2026-08-18T20:13:21.543938Z","submitted_at":"2023-12-30T04:06:16Z","title":"Unicron: Economizing Self-Healing LLM Training at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.00134","snapshot_observed_at":"2026-08-15T20:33:37.515264Z","title":"Unicron: Economizing self-healing llm training at scale,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.515264Z"},"links":{"cited_paper":"/paper/2401.00134","citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:928d3808d4ef0e911b30e7766b137d3137e426285151984cf09c8e1eae10cb89","observation_id":"ab74ce4c-e418-49d0-9e33-f05a6525cc69","resolution":{"observed_at":"2026-08-15T20:33:37.515264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.477032Z","title":"Datastates-llm: Lazy asynchronous checkpointing for large language models,","venue":null,"work_id":"0cdb6fc3-ef2b-4243-8424-8ab42da700eb","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.519570Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:71573f748c0192be38267d884eb66da0fc066cb2b8f73752125b287b71306909","observation_id":"3a08cafc-3df1-4c5b-a1f2-5e5c1172d5ee","resolution":{"observed_at":"2026-08-15T20:33:38.539534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.466560Z","title":"Checkfreq: Frequent, fine-grained dnn checkpointing,","venue":null,"work_id":"465e093f-7fa2-449c-ad5e-659862b25ce5","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.522539Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:09c81748046fa34b55c0b001ce3b896093a043e997fff877766712cb0b01c50c","observation_id":"a8fc9856-707a-4456-8545-06a713169783","resolution":{"observed_at":"2026-08-15T20:33:38.470657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.456906Z","title":"Just-in-time checkpointing: Low cost error recovery from deep learning training failures,","venue":null,"work_id":"8c492944-bc6f-48b3-940e-99b204ff7b8c","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.526113Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:4428d082b97dea0fea928495d4b73c60b90457ac6e406f313224f4e16d87403d","observation_id":"225a613b-7547-414a-bb5d-e2133ebae9d1","resolution":{"observed_at":"2026-08-15T20:33:38.460222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.446625Z","title":"Check-n-run: A check- pointing system for training deep learning recommendation models,","venue":null,"work_id":"04f7d0de-029a-473a-bf70-3a1afed39371","year":2022},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.529860Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:f8a764df7d59d33d1d54c017e90b4bee657e9a566879eb373357cd985dc122a0","observation_id":"188bca24-08c5-41ca-bfcc-3e43aa74d2a8","resolution":{"observed_at":"2026-08-15T20:33:38.450394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.437780Z","title":"Gemini: Fast failure recovery in distributed training with in-memory checkpoints,","venue":null,"work_id":"4a7fca47-268e-45d8-ab12-ef8a5e89b73c","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.533077Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:d24aae146f5ada29a69eb186114790e396f8187ba97f24af6d70ab5a24491fb3","observation_id":"c7276eb5-cd0d-49fe-8bef-2ebbb8f59d15","resolution":{"observed_at":"2026-08-15T20:33:38.440994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.428199Z","title":"Pollux: Co-adaptive clus- ter scheduling for goodput-optimized deep learning,","venue":null,"work_id":"6bf0f94b-f140-47a0-afdd-4037728edb31","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.536294Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:fdcbb9ed42aa4b94ae1f7b5df1a8ba9ecb388199600a81fe7bcba5348530cdfb","observation_id":"50c58506-781e-4b12-8fff-802b626f2a15","resolution":{"observed_at":"2026-08-15T20:33:38.431496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08586","last_updated":"2024-08-16T07:41:52Z","snapshot_observed_at":"2026-08-16T13:26:05.261522Z","submitted_at":"2024-08-16T07:41:52Z","title":"Rubick: Exploiting Job Reconfigurability for Deep Learning Cluster Scheduling","version":1},"cited_work":{"arxiv_id":"2408.08586","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.08586","snapshot_observed_at":"2026-08-15T20:33:37.969236Z","title":"Rubick: Exploiting Job Reconfigurability for Deep Learning Cluster Scheduling","venue":"cs.DC","work_id":"977078da-7f70-4e3f-8002-55ef3f6f1e21","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.539956Z"},"links":{"cited_paper":"/paper/2408.08586","citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:b09807e2c2accba2cc72c75924db6a56126c2c4fa2ac3cea8890320ee2b3839e","observation_id":"73c5ee27-5fa4-4085-bb94-c8b89fceb1e1","resolution":{"observed_at":"2026-08-15T20:33:37.973013Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.416162Z","title":"Optimus: an efficient dynamic resource scheduler for deep learning clusters,","venue":null,"work_id":"e24fdc2e-5fb4-492f-8eba-3a7f221bb4e0","year":2018},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.543896Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:fcf6dd1883dba211ea79b31aafd5ab940b5eca19002c1a42aa1c062459064809","observation_id":"f2e16fa9-434d-4704-94dd-168520aac318","resolution":{"observed_at":"2026-08-15T20:33:38.421195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.405310Z","title":"Deepboot: Dynamic scheduling system for training and inference deep learning tasks in gpu cluster,","venue":null,"work_id":"b578f9c5-0457-4551-a978-ed6c4885c7ca","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.547109Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:1294896508aab93933516dd3b018074dc886b20b6910b4d5d0f62dda27024e25","observation_id":"dd8094ff-9c74-4b64-8074-6502e00354bf","resolution":{"observed_at":"2026-08-15T20:33:38.409460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.395140Z","title":"Universal checkpointing: A flexible and efficient distributed checkpointing system for large-scale dnn training with reconfigurable parallelism,","venue":null,"work_id":"d3b4ccbc-5dbc-4f21-9155-79d19de19b5b","year":2025},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.550543Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:8e31a423ddd64a8675c48989a3119e475a09f788176ee9baabe650f7f053f46f","observation_id":"cc693fd2-d382-43dc-9658-875cfdf69a43","resolution":{"observed_at":"2026-08-15T20:33:38.398690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.384715Z","title":"Paddlepaddle: An open-source deep learning platform from industrial practice,","venue":null,"work_id":"5260ade2-46cf-4482-be89-36bb7ea3f81e","year":2019},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.553772Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:1b84818998980594391f57c7ba8ac28e57968ca79e21aeea9c252323429ded57","observation_id":"a5433776-98d7-491f-8c17-624036ac160f","resolution":{"observed_at":"2026-08-15T20:33:38.388263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.328306Z","title":"Elasticdl: A kubernetes-native deep learning framework with fault-tolerance and elastic scheduling,","venue":null,"work_id":"49949866-639f-4ec2-b20f-fc629d640440","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.556767Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:7e63cdd01ab1aa0d31fcccf5d74bd671e5602b1c955f8a842e8d4cfb5097984c","observation_id":"79b70c3d-3d72-4b91-9724-bf47b3af1e17","resolution":{"observed_at":"2026-08-15T20:33:38.363091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.282954Z","title":"Dl2: A deep learning-driven scheduler for deep learning clusters,","venue":null,"work_id":"05c6ac19-27cd-4408-aa48-74cb80cb2da4","year":1947},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.559492Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:1a49ec7ccfb9b58214b1726f7d5f36ad0cd9c06cf35d8a45fe14cee3edf56e8c","observation_id":"a38c8d34-5691-45a6-a8be-39bf093abafc","resolution":{"observed_at":"2026-08-15T20:33:38.305471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.251820Z","title":"Elastic parameter server: Accelerating ml training with scalable resource scheduling,","venue":null,"work_id":"ce4e725e-4ac8-4811-b42c-81b90fc6c34e","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.608254Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:9c7a7fa3241ac31b76fb5080ba7f95977032484ddea7dfc4f7062912e4a37251","observation_id":"6b11ba66-8093-407a-a292-5aa028b3e40a","resolution":{"observed_at":"2026-08-15T20:33:38.255361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.240870Z","title":"Elastic deep learning in multi-tenant gpu clusters,","venue":null,"work_id":"756f8b77-dfcd-4607-9936-e2a9760832dd","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.656371Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:1f76b731fd5e01e76f950b89e699715cea151ffff61bde9bcdebea01c038b927","observation_id":"ae7f6daa-ad87-457c-8b07-d4d0349d2b97","resolution":{"observed_at":"2026-08-15T20:33:38.244563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.230509Z","title":"Elastic resource sharing for distributed deep learning,","venue":null,"work_id":"89255932-cc8f-4ece-a271-13827115a41f","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.745932Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:53e4cabbf047809020a62c60ac895848d4722410ee481b9919f3b05aa8300dbf","observation_id":"019a71cb-9e98-43fb-845b-2ffa93fb7a0d","resolution":{"observed_at":"2026-08-15T20:33:38.234311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.219938Z","title":"Elan: Towards generic and efficient elastic training for deep learning,","venue":null,"work_id":"d6df981d-6df2-424f-99a6-ab333992ea97","year":2020},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.857839Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:b8134270e6cfaadcafa6a7a28347440b409eac50fdf0ea26ba9149a20a6947bc","observation_id":"9ca9be8e-ef1a-47d1-9104-2fb17b682493","resolution":{"observed_at":"2026-08-15T20:33:38.223437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.208142Z","title":"Resource elasticity in distributed deep learning,","venue":null,"work_id":"425af1ca-8ff2-4551-a925-648a674f1e9c","year":2020},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.900940Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:48594245ab4b0a7d44ab6e2f131c37e8b063741b7b933b2b72db6cdcf16d71e1","observation_id":"b971d182-f0d1-4e26-959d-8408d17a3d9b","resolution":{"observed_at":"2026-08-15T20:33:38.213215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.185504Z","title":"Comparing decentralized learning to federated learning when training deep neural networks under churn,","venue":null,"work_id":"809e0c44-886f-4eb8-a02a-44ec3f157a30","year":2021},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.904626Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:b9b983b6f57469aaddf141fd5394352c7ef1e600354efd405c109dbc84172913","observation_id":"6f5478f8-0000-4ccf-99cd-f27d71ecc3bd","resolution":{"observed_at":"2026-08-15T20:33:38.200771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05662","last_updated":"2026-06-08T03:58:45Z","snapshot_observed_at":"2026-08-16T13:11:40.487327Z","submitted_at":"2024-10-08T03:22:14Z","title":"Communication-Efficient Federated Learning under Dynamic Device Arrival and Departure: Convergence Analysis and Algorithm Design","version":4},"cited_work":{"arxiv_id":"2410.05662","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05662","snapshot_observed_at":"2026-08-15T20:33:37.953293Z","title":"Communication-Efficient Federated Learning under Dynamic Device Arrival and Departure: Convergence Analysis and Algorithm Design","venue":"cs.LG","work_id":"8b42cce4-db11-43d6-8006-79012362e254","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.907767Z"},"links":{"cited_paper":"/paper/2410.05662","citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:254d2b521060730bb3de5efc85561f7218dfd8dce7af52173651d9fff7517793","observation_id":"caa64a81-5a90-418a-baf6-b2e26d344caa","resolution":{"observed_at":"2026-08-15T20:33:37.958678Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.075512Z","title":"Mimic: Combating client dropouts in federated learning by mimicking central updates,","venue":null,"work_id":"23ac5a72-c078-465e-a7b7-38b95574e7f0","year":2023},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.911965Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:34e69caf0e65c4659d1cd5c76c58d463f87e9b238627c9cde7baac40c8e0cf9f","observation_id":"ce4b3a87-1b3a-4af4-accc-bb63d165e478","resolution":{"observed_at":"2026-08-15T20:33:38.133480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.021669Z","title":"Esync: Accelerating intra-domain federated learning in heterogeneous data centers,","venue":null,"work_id":"b51154a4-3f14-4aa8-b0e0-93bf9fad6899","year":2020},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.915291Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:36285a0f86b26dd9a472af1fce4c2fe42d72d8f0afc6a186ce5333b764718043","observation_id":"c2c13fb0-6c94-4780-a39c-13bb7e8257f3","resolution":{"observed_at":"2026-08-15T20:33:38.025439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.010643Z","title":"Accelerating geo-distributed ma- chine learning with network-aware adaptive tree and auxiliary route,","venue":null,"work_id":"87abde07-e4a2-46b0-8faa-012514c3fca6","year":2024},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.918473Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:6c480ab715805918ac14d5834cf0c0030b859c9a7a0c43e33513c6e421abcbc8","observation_id":"d3a14cc4-d201-4277-b26a-1ba635be60e7","resolution":{"observed_at":"2026-08-15T20:33:38.015398Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:38.000069Z","title":"Data heterogeneity-robust federated learning via group client selection in industrial iot,","venue":null,"work_id":"91dacf03-682c-4015-abdf-d517a4f7e433","year":2022},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.922857Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:0e4646b522efa5d9cbe0144a2a5ea6bd6746ba291a40ba19f9a475ca03e92109","observation_id":"f9883d22-ef01-41b6-ab94-8cdb9aab66e1","resolution":{"observed_at":"2026-08-15T20:33:38.004206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:33:37.989547Z","title":"The art of computer programming,","venue":null,"work_id":"60530a67-aa47-408f-91d1-27ca9d4f8cec","year":1999},"citing_paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T20:33:37.925834Z"},"links":{"citing_paper":"/paper/2505.12815"},"observation_digest":"sha256:1d9889867ac4cb7e0e3d40ac55f8b043b089f841974a24066e090cd911141b4d","observation_id":"6e8d9a27-fc0f-4bd5-805a-a013ffab0427","resolution":{"observed_at":"2026-08-15T20:33:37.992980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.12815","last_updated":"2025-09-13T18:39:29Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-16T18:22:42.312267Z","submitted_at":"2025-05-19T07:52:17Z","title":"Learning In Chaos: Efficient Autoscaling and Self-Healing for Multi-Party Distributed Training"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":1,"verified_exact":2,"verified_fuzzy":35},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2505.12815."}