{"as_of":"2026-08-10T21:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5dd51c4e7fd2986050a90f9a7fe343078231425bf776cfff0998a598bce604e0","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T11:01:20.775942Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.03407/citation-record","integrity":"/paper/2509.03407/integrity","json":"/paper/2509.03407/citation-record.json","paper":"/paper/2509.03407"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.254394Z","title":"Jiang, F.F","venue":null,"work_id":"f89a1720-57b7-48a7-b660-1518c2617226","year":2020},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.253637Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:7bd3c87fcfc8bf73507d44209a10316823fc2d15a1a87e8d9fc17dc26801da90","observation_id":"369b2bc1-0d80-4a3a-97ad-37580e6c3fbd","resolution":{"observed_at":"2026-08-05T11:01:23.269785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.01066","last_updated":"2019-09-04T09:33:20Z","snapshot_observed_at":"2026-07-06T08:18:37.267833Z","submitted_at":"2019-09-03T11:11:08Z","title":"Language Models as Knowledge Bases?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.01066","snapshot_observed_at":"2026-08-05T11:01:20.264626Z","title":"Petroni, T","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.264626Z"},"links":{"cited_paper":"/paper/1909.01066","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:eacbb67c226f1b6b1c283f4a33be4eb37c516c077ee823c007cc52bee20f96a2","observation_id":"3c0caf28-f680-4520-872a-bc7a9b3bf304","resolution":{"observed_at":"2026-08-05T11:01:20.264626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.209269Z","title":"Vaswani, N","venue":null,"work_id":"1d35f681-51f7-4b24-bc00-914d01970c26","year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.279652Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:6365e69fb5724ac2312748a97750ba796c75e9a4b702554fa5fc59cbba21dd95","observation_id":"3f3b7abb-4f89-4bdd-9016-6773aeda747d","resolution":{"observed_at":"2026-08-05T11:01:23.216221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.14871","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:21.877434Z","title":"Gross, Y","venue":null,"work_id":"69937e27-54d9-4f8c-8a98-210d626a287a","year":2025},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.286016Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:8088f6e7941ba4542c0dbf0f73c21359a998467fb4db6394fe3f4b1cb95b2ec9","observation_id":"37b64c2d-e083-4111-83ea-fba4661a5b93","resolution":{"observed_at":"2026-08-05T11:01:21.892836Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.175334Z","title":"Lu, Full-text federated search in peer-to-peer networks, 2007","venue":null,"work_id":"2dae593e-1869-45b5-8d28-7927de57843a","year":2007},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.292943Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:748326dff5adee67aa1c5b19c5d9334177af6ed9756e1b7b93394cc7f09e3763","observation_id":"efea2470-27a0-4874-8a5d-715fe1b0bba4","resolution":{"observed_at":"2026-08-05T11:01:23.183583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.140356Z","title":"Chang, X","venue":null,"work_id":"9f05eceb-6cfd-4740-a561-ed7d4946f4ff","year":2024},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.298954Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:73bdc5ea0f12030864757a0e262318b04b538f7177653edcd3962aca2787ead9","observation_id":"2d91ef00-4631-4963-b6b4-2be282477bd8","resolution":{"observed_at":"2026-08-05T11:01:23.148909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T11:01:20.305515Z","title":"Mielke, Z","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.305515Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:c5299d66758b105e0f7346a26a4928285f2a0ce23d2a8b2b6cb821fcc3a71340","observation_id":"abb38b1c-7949-4d7b-94ac-cfc58ebfb653","resolution":{"observed_at":"2026-08-05T11:01:20.305515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.109411Z","title":null,"venue":null,"work_id":"e872dcad-49a2-40fa-85cd-80956b3e9106","year":2023},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.313984Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:97cf0f2c717c8be4cc978ec2956a96cff171c3a42a59985c52ac3336e0e52c27","observation_id":"511fed35-b51e-4dc3-a5d2-04d253d9f0f5","resolution":{"observed_at":"2026-08-05T11:01:23.115324Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.074131Z","title":null,"venue":null,"work_id":"210b73b1-c7d1-4ad2-9436-6aaa879c63f8","year":2020},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.319527Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:9dae5b3676521421071d594c4c50df88a09a4186dfad25dffba69eca634e2fa7","observation_id":"36ecef3a-31c2-4366-bad0-cbd67d428486","resolution":{"observed_at":"2026-08-05T11:01:23.086616Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.048356Z","title":"Mesnil, Y","venue":null,"work_id":"2e800465-14af-4540-a535-8df8096d8482","year":2012},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.325294Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:40d5a2529e415c505a0926ee2b364d185359da576e9211c3fa4d075f09ef7719","observation_id":"b18a4e04-00ee-49d4-ac99-3a4e366ab7a9","resolution":{"observed_at":"2026-08-05T11:01:23.054290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:23.019138Z","title":"Yosinski, J","venue":null,"work_id":"1b32c1e0-35ff-4351-93d7-09d5ff9564f9","year":2014},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.331108Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:075ed3cdaa3fd1502c5bb82c8d2f71cebb5186f2578447fc5adc1e2913a7b389","observation_id":"f7c19671-fb29-4c3e-836f-93612b89b328","resolution":{"observed_at":"2026-08-05T11:01:23.028909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.994792Z","title":"Lozano-Diez, O","venue":null,"work_id":"1b197173-6409-44b1-9006-b556fc3568d9","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.336783Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:3e70e373efab5e8b87b9987908324e7630fba195bbb2f1156c3bf85ebe60fb19","observation_id":"0d108b21-501d-4bd9-aab2-54bac5bc3069","resolution":{"observed_at":"2026-08-05T11:01:23.000216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.969570Z","title":null,"venue":null,"work_id":"53189c13-f147-47ec-a6da-00db6d4c3ab5","year":2016},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.342251Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:5269af36630a8d08cd98829dd1dd9ba4d3497d3dfccd0d15e25b2b195810c881","observation_id":"277a9e2a-8438-4b98-98f7-960f9f8f00ce","resolution":{"observed_at":"2026-08-05T11:01:22.981228Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.937125Z","title":"LeCun, K","venue":null,"work_id":"bf51b6a1-67a1-4576-82e9-0d72b4409564","year":2010},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.351289Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:223196c38895027267b6b1be0fcf381654e7f3bedad8af3ca019cce32f5c896d","observation_id":"f1375fff-70cf-4580-9526-bd6c83531a65","resolution":{"observed_at":"2026-08-05T11:01:22.947415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.909867Z","title":"Britz, Understanding convolutional neural networks for NLP, Denny’s Blog, (2015)","venue":null,"work_id":"d643fd66-d472-4cf5-a18a-8905c48bb361","year":2015},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.358207Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:d1a09c8bdd970c73fe83261a4d40303fe8b2a496e55fbd8297ad74154511b581","observation_id":"573cd345-cf2e-466b-910f-43a0d6d5cc27","resolution":{"observed_at":"2026-08-05T11:01:22.921786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05704","last_updated":"2022-06-07T19:25:30Z","snapshot_observed_at":"2026-08-10T03:12:27.237826Z","submitted_at":"2021-04-12T17:58:56Z","title":"Escaping the Big Data Paradigm with Compact Transformers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.05704","snapshot_observed_at":"2026-08-05T11:01:20.370715Z","title":"Hassani, S","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.370715Z"},"links":{"cited_paper":"/paper/2104.05704","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:91e041b27bb36b0f715f36d88527d751810f88f3d13c101030f2cb5741b35d63","observation_id":"6fabdab6-57d1-4bee-b322-4008fc0d6c37","resolution":{"observed_at":"2026-08-05T11:01:20.370715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12900","last_updated":"2025-04-09T13:06:49Z","snapshot_observed_at":"2026-08-10T16:36:42.300446Z","submitted_at":"2025-01-22T14:19:48Z","title":"Unified CNNs and transformers underlying learning mechanism reveals multi-head attention modus vivendi","version":3},"cited_work":{"arxiv_id":"2501.12900","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.12900","snapshot_observed_at":"2026-08-05T11:01:21.451284Z","title":"Unified CNNs and transformers underlying learning mechanism reveals multi-head attention modus vivendi","venue":"cs.LG","work_id":"1de1d63f-d349-4112-a364-f6495fffc640","year":2025},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.380285Z"},"links":{"cited_paper":"/paper/2501.12900","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:a71c6da0229d915d17f45f4b0a42c263ff05d984868f1e6e570d8d04851e2b78","observation_id":"1ed22dd2-4185-4685-9fd0-c9d8a683b686","resolution":{"observed_at":"2026-08-05T11:01:21.466027Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T11:01:20.387468Z","title":"Dosovitskiy, L","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.387468Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:fd696ec540ec828861854408b1832d2e561e4fbc51a7c6eaff0489f56dc1faa2","observation_id":"3841cc74-b2b7-4b5c-878b-01cd924c4258","resolution":{"observed_at":"2026-08-05T11:01:20.387468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23832","last_updated":"2025-06-30T13:23:46Z","snapshot_observed_at":"2026-08-10T03:13:46.347027Z","submitted_at":"2025-06-30T13:23:46Z","title":"Low-latency vision transformers via large-scale multi-head attention","version":1},"cited_work":{"arxiv_id":"2506.23832","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23832","snapshot_observed_at":"2026-08-05T11:01:21.358336Z","title":"Low-latency vision transformers via large-scale multi-head attention","venue":"cs.CV","work_id":"a80be56d-6dbe-43e8-a384-dee94ae7c37e","year":2025},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.394734Z"},"links":{"cited_paper":"/paper/2506.23832","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:92edaac02fe30ecf4d76e7e71eb0e08d5fea63f3fc7539aa8ecf48966f02ae97","observation_id":"15c01bdf-e0a9-440b-a0c5-a8f1d4c37ffb","resolution":{"observed_at":"2026-08-05T11:01:21.365202Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.886163Z","title":"Jawahar, B","venue":null,"work_id":"d1bd6daf-c9f1-4880-8c2b-8a1035ce23fd","year":2019},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.402323Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:af3a27f381fdd96c74b42b85799d6c08a6a7d8f73795a5467905b9dcc3baa35a","observation_id":"03ae86fe-23f8-4480-97f4-cee96930a274","resolution":{"observed_at":"2026-08-05T11:01:22.893302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.04341","last_updated":"2019-06-11T01:31:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-06-11T01:31:41Z","title":"What Does BERT Look At? An Analysis of BERT's Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.04341","snapshot_observed_at":"2026-08-05T11:01:20.413311Z","title":"Clark, U","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.413311Z"},"links":{"cited_paper":"/paper/1906.04341","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:e2d6e27e9c2d495b09e566fe1240f4079e504f98f82e6873fea3dc7b726b5db8","observation_id":"382a984d-f56c-418c-b26b-baa5b667b289","resolution":{"observed_at":"2026-08-05T11:01:20.413311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.859000Z","title":"Rogers, O","venue":null,"work_id":"3d4d8d4b-cf29-455d-a9ae-0b4e22517fda","year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.419902Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:fd6a8e98555b5cccaf67cad9a8188069cf183fa4fe3b62c2f093873cd7bf54ee","observation_id":"912697f5-0a22-46a5-9e87-fe461e15f8ad","resolution":{"observed_at":"2026-08-05T11:01:22.867005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.831362Z","title":"Devlin, M.-W","venue":null,"work_id":"316b5945-e261-46d5-ab7d-084cd8ec5939","year":2019},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.430206Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:a9cd4d81b4e046c10e517a05864eb8f8e17f5cf2981eb65995d9800e4dfc49b3","observation_id":"62af66e6-b51c-46d9-8a30-8c6fd80b9791","resolution":{"observed_at":"2026-08-05T11:01:22.839704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.800704Z","title":"Hoshen, R","venue":null,"work_id":"b823128f-cde2-4e17-a5bc-ae9a9ed6837c","year":1976},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.438559Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:c1daae965937ab79ce3e2a910ae62712cd8dffedf01c942047f00ea4f1e20053","observation_id":"70b8e495-7f6c-4b91-9cd3-8d9f2c248f82","resolution":{"observed_at":"2026-08-05T11:01:22.807136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.773364Z","title":"Havlin, R","venue":null,"work_id":"d62ff9a1-7bc0-4623-b75e-e8ee5169b151","year":1984},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.445365Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:d4a508d08b9c11225d2fa0a238ef666ddabfb913c41f3c14b48be0d7fc12c008","observation_id":"d5b3536b-e0c7-4857-bf2f-06a3f49b6e3c","resolution":{"observed_at":"2026-08-05T11:01:22.781412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.746003Z","title":null,"venue":null,"work_id":"43defb52-3af9-42d1-8019-cb49c19fd3a9","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.455850Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:ceb7ce1d497be74f0d18003be7953b4e021972db422b1022c3eb78093aa21cd1","observation_id":"a770c096-19e6-4e4f-ad90-99a89a0b4bc6","resolution":{"observed_at":"2026-08-05T11:01:22.753787Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.718557Z","title":null,"venue":null,"work_id":"36669e6b-e616-4a31-a1e7-01b704450685","year":2020},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.464714Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:43565607692dd57697c8be196a622c9d2ca6af40e220ccfcd3c21caceb5d97a1","observation_id":"f9da1e96-a30d-4112-9b6d-083a0f43ff63","resolution":{"observed_at":"2026-08-05T11:01:22.726454Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.00512","last_updated":"2019-09-02T01:51:46Z","snapshot_observed_at":"2026-08-09T14:44:59.104271Z","submitted_at":"2019-09-02T01:51:46Z","title":"How Contextual are Contextualized Word Representations? Comparing the Geometry of BERT, ELMo, and GPT-2 Embeddings","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.00512","snapshot_observed_at":"2026-08-05T11:01:20.475598Z","title":"Ethayarajh, How contextual are contextualized word representations? Comparing the geometry of BERT, ELMo, and GPT-2 embeddings, arXiv preprint arXiv:1909.00512, (2019)","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.475598Z"},"links":{"cited_paper":"/paper/1909.00512","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:336d345d6433931e13f70eccf0ba215ff5e3c7f034eba9d994bdadae61e99e0f","observation_id":"333fe989-c5d2-4aea-9789-89d3b212612c","resolution":{"observed_at":"2026-08-05T11:01:20.475598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1301.3781","last_updated":"2013-09-07T00:30:40Z","snapshot_observed_at":"2026-07-06T03:04:11.148340Z","submitted_at":"2013-01-16T18:24:43Z","title":"Efficient Estimation of Word Representations in Vector Space","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1301.3781","snapshot_observed_at":"2026-08-05T11:01:20.487297Z","title":"Mikolov, K","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.487297Z"},"links":{"cited_paper":"/paper/1301.3781","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:1e19494203278a4be5f00ec00126348203cf4a2b19827713f1a002bba37c049d","observation_id":"be0ba031-8915-4bc2-86bc-b4538dfb9263","resolution":{"observed_at":"2026-08-05T11:01:20.487297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.692244Z","title":"Radovanovic, A","venue":null,"work_id":"293e2ed0-a051-4d92-934c-ac8b1a8b7baa","year":2010},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.491991Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:216dbaa2987fa25a24786082da21d7dfed877c871d429a46ff57ed892c8139ca","observation_id":"6a4816fd-5420-4eb0-979a-362a7f3ccebd","resolution":{"observed_at":"2026-08-05T11:01:22.700400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1702.01417","last_updated":"2018-03-19T20:28:27Z","snapshot_observed_at":"2026-07-06T05:28:51.934048Z","submitted_at":"2017-02-05T15:43:07Z","title":"All-but-the-Top: Simple and Effective Postprocessing for Word Representations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1702.01417","snapshot_observed_at":"2026-08-05T11:01:20.498553Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.498553Z"},"links":{"cited_paper":"/paper/1702.01417","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:1c9d53d3a71fef48f550c5aa435e416147662a0c661b039a79783c4abee07315","observation_id":"2276999c-2920-411c-9467-7afc4400f44b","resolution":{"observed_at":"2026-08-05T11:01:20.498553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1710.04087","last_updated":"2018-01-30T14:41:51Z","snapshot_observed_at":"2026-08-10T11:57:59.118684Z","submitted_at":"2017-10-11T14:24:28Z","title":"Word Translation Without Parallel Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.04087","snapshot_observed_at":"2026-08-05T11:01:20.506555Z","title":"Conneau, G","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.506555Z"},"links":{"cited_paper":"/paper/1710.04087","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:3ce2f02b3eee9663fb6cbc93abc3904ff51b10ba5fb4271fae46d4da16bab0df","observation_id":"d01e408a-5275-4a4a-93ab-a95cc3ba023c","resolution":{"observed_at":"2026-08-05T11:01:20.506555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.652555Z","title":"Dhillon, D.S","venue":null,"work_id":"71fad85f-9447-4b88-8f9c-c3f889798276","year":2001},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.512982Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:a92be7ef271063e4b9cec8b7ec3f7e845e82982decf915f4d85881063737e914","observation_id":"21183661-eaf4-4357-a91a-44ec5b6c5cdc","resolution":{"observed_at":"2026-08-05T11:01:22.664750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.621097Z","title":"Banerjee, I.S","venue":null,"work_id":"82771c15-45b7-4c87-9c96-8df7955400f8","year":2005},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.519479Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:c48a71b00fdbfbf9b9bdca72375090c8fec6e0487f571002fdfdc56c9aecb797","observation_id":"79b0c558-c92c-4426-8ea4-2c11bc43c403","resolution":{"observed_at":"2026-08-05T11:01:22.629630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.583203Z","title":"McInnes, J","venue":null,"work_id":"f4b74768-5e63-46bf-976b-acfd90c5404d","year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.525200Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:3a265b775f9a7f3b332adc810a25da07f7dba49885287957d8271ed635aca01a","observation_id":"a6f9bbaa-9ee5-4a7b-b4b4-766c361ef0e2","resolution":{"observed_at":"2026-08-05T11:01:22.590187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.555629Z","title":"Pratap, A","venue":null,"work_id":"82f28b81-f5dd-4f13-8258-7379aa364a7a","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.532933Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:9f5a331c6c3c0160bac706cedf970cb5d3da1c1ee0d9639cb30b5bec5fa5bae2","observation_id":"579bb205-9f6e-46c9-b327-b768610a0455","resolution":{"observed_at":"2026-08-05T11:01:22.562796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.520055Z","title":"Faraki, X","venue":null,"work_id":"40d38386-b1e1-4c3f-9dd6-a147bb5373a1","year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.539945Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:a95ac665e869b70afb7c1037097f1f27083972c21d8a5db99495b3b9ca29b44d","observation_id":"eb0cd428-4523-4682-9d74-64d5b43c57b2","resolution":{"observed_at":"2026-08-05T11:01:22.529302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.495896Z","title":"Tzach, Y","venue":null,"work_id":"134c8e50-797d-4165-af18-31563b9a4cdc","year":2025},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.547256Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:757f4685ccf7012e703aacb7fb405262f344507677bcc5f7b0c04c5b74cb30ae","observation_id":"ab9eb131-308c-477b-a8d1-b81cf64c094f","resolution":{"observed_at":"2026-08-05T11:01:22.504677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.469843Z","title":null,"venue":null,"work_id":"672b1834-2f0d-4485-99ad-d60fa71588d1","year":2024},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.561970Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:73e6b1eb74f28d6f58ce314466df8fa62f96494b6681a2270a62ddeedfadb4b9","observation_id":"17de94ba-393a-459a-97ea-5a4c1fad46c8","resolution":{"observed_at":"2026-08-05T11:01:22.476441Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18078","last_updated":"2023-05-29T13:28:43Z","snapshot_observed_at":"2026-08-10T02:48:47.955267Z","submitted_at":"2023-05-29T13:28:43Z","title":"The mechanism underlying successful deep learning","version":1},"cited_work":{"arxiv_id":"2305.18078","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.18078","snapshot_observed_at":"2026-08-05T11:01:21.187741Z","title":"The mechanism underlying successful deep learning","venue":"cs.CV","work_id":"a456e3e3-102a-4394-8366-e2651aa53a92","year":2023},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.570838Z"},"links":{"cited_paper":"/paper/2305.18078","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:2dae09e4b6320863c2d9fd2bdeccda40b3614d86854aa5c4c151a2cb80d9ef28","observation_id":"f1765c35-a044-4908-b3a0-71d8604a444e","resolution":{"observed_at":"2026-08-05T11:01:21.194994Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01781","last_updated":"2017-01-27T12:49:11Z","snapshot_observed_at":"2026-07-06T04:58:57.726406Z","submitted_at":"2016-06-06T15:14:50Z","title":"Very Deep Convolutional Networks for Text Classification","version":2},"cited_work":{"arxiv_id":"1606.01781","doi":null,"metadata_source":"pith","pith_arxiv_id":"1606.01781","snapshot_observed_at":"2026-08-05T11:01:21.141239Z","title":"Very Deep Convolutional Networks for Text Classification","venue":"cs.CL","work_id":"d14874d3-6f78-42b7-b334-7b1065998c77","year":2016},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.579520Z"},"links":{"cited_paper":"/paper/1606.01781","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:a697862a8a9168190e5438953080738fa911e636f43f90ce76e37494d5d5d038","observation_id":"80bbdf13-0727-4112-843e-46dcb890fbeb","resolution":{"observed_at":"2026-08-05T11:01:21.156449Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.438484Z","title":"Krizhevsky, G","venue":null,"work_id":"19b9fa42-d179-4ae0-b343-2db47f8bce62","year":2009},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.590749Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:5e1b6d8cdf7f7ff277cefe715d2e730c75549aea3348e7d1ee0955df7e6ec1e5","observation_id":"c9e7a37f-fa66-49f7-89c4-8a220ff45b81","resolution":{"observed_at":"2026-08-05T11:01:22.448658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.402446Z","title":null,"venue":null,"work_id":"f2d058ee-a7f5-4afd-89da-3fd946b639d8","year":2023},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.600503Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:51ffe7270666112c9913c50cac7e8b49b79f3bf108b1de74aec4c1733b261d19","observation_id":"e4025a3c-95c4-4b99-b04e-670fc6625b99","resolution":{"observed_at":"2026-08-05T11:01:22.411258Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.378557Z","title":"Koresh, T","venue":null,"work_id":"1156d153-e08c-4989-b6d0-d64ea2a49a78","year":2024},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.606967Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:29615db72698fe266851edd7af7d3b4f4cbb6db4aebb8d6723807ca211b7eb88","observation_id":"19729283-6321-49d6-ae50-bdeb80f6d3e8","resolution":{"observed_at":"2026-08-05T11:01:22.384467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.352312Z","title":"Tevet, R.D","venue":null,"work_id":"c42b1874-7bd9-402f-8b88-4fb879f93e46","year":2024},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.612027Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:7b31fa52a2585d8081d461d27bfee906a03f51d2e8282cdc36bfc23cd6128849","observation_id":"fd4e3301-02dc-4c27-9d40-77839a5cdca7","resolution":{"observed_at":"2026-08-05T11:01:22.358518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07537","last_updated":"2024-03-12T10:46:33Z","snapshot_observed_at":"2026-08-09T07:29:19.203455Z","submitted_at":"2023-09-14T09:03:57Z","title":"Towards a universal mechanism for successful deep learning","version":2},"cited_work":{"arxiv_id":"2309.07537","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.07537","snapshot_observed_at":"2026-08-05T11:01:21.084697Z","title":"Towards a universal mechanism for successful deep learning","venue":"cs.CV","work_id":"1dc62d5a-33d0-4946-ab3a-716d2f85975a","year":2023},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.617951Z"},"links":{"cited_paper":"/paper/2309.07537","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:1aff63dd21c3822a7654f53b7cce451070d410b356c943a4eff2419b03b6d426","observation_id":"92d1ca76-cc54-4e41-9258-5ad28b2aa532","resolution":{"observed_at":"2026-08-05T11:01:21.095075Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.325967Z","title":null,"venue":null,"work_id":"14e11944-07de-4655-a0d1-ab4e0b244b15","year":2020},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.625146Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:5d03244ba578b6c85c98d8cdcb2fd4d0b9343bbf93a4f7a28ab323c5295935ab","observation_id":"193d0cc9-17f3-4aea-9bf0-6645984e830f","resolution":{"observed_at":"2026-08-05T11:01:22.331304Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.285988Z","title":null,"venue":null,"work_id":"a5df7edc-b7ea-407d-a86b-5c5b1b4a00ce","year":2020},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.634231Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:3ff7acb95c80422d7eebbdc5900100bc278e47689d956060066e14e16c9c3e07","observation_id":"d6082b6c-0143-4185-963b-7f4715270e67","resolution":{"observed_at":"2026-08-05T11:01:22.295675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.250263Z","title":"Clark, K.A","venue":null,"work_id":"0f1ca20a-03d1-46c5-9406-8bc53946f188","year":1999},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.649209Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:0c07bc7cdb21c4deb636ee05c16bc33aed0a1fec034f7a30216ce89f1066854e","observation_id":"88972625-d791-42bb-8c9d-812301949b95","resolution":{"observed_at":"2026-08-05T11:01:22.260894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.219466Z","title":"Ghosh, Y","venue":null,"work_id":"af50ccf7-8bc5-4205-84e4-0d28fdf9a53b","year":1992},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.658372Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:7c7986fca487184c1cd8ad34e4f586ac5bff38fa3f5d447ac8bd6aa2e085b86a","observation_id":"3100a1d0-313b-4774-8ed5-e39a1305eed7","resolution":{"observed_at":"2026-08-05T11:01:22.225720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.190968Z","title":"Durbin, D.E","venue":null,"work_id":"464df067-4396-4443-ac6d-425c3ec687ad","year":1989},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.669557Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:6210a8b86d17bf2bee8ebc1785d5564a6bc7ebec5915ed34506a85bb66d73b7b","observation_id":"d8eda53d-4216-4a5b-bd20-b072626b2485","resolution":{"observed_at":"2026-08-05T11:01:22.197814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.161940Z","title":"Hodassman, R","venue":null,"work_id":"fef48a8f-3df0-4dbe-86fc-88e3ae26b8d7","year":2022},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.676882Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:e6af516b0952431310f749c24ecb590e40e19aba0a7f0c4fbc5226c3ec33a16c","observation_id":"8fe21178-1646-4e52-8699-85ce366fdb1f","resolution":{"observed_at":"2026-08-05T11:01:22.169676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.126418Z","title":"Vardi, Y","venue":null,"work_id":"ff3f8a26-b366-4cc7-8fcd-45b689add10d","year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.683639Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:8ca42ce5522c3702fa29d8ff3be7dbea9f528bcb92a3d7a7ceee61865416fb35","observation_id":"c52c2b6f-4cb3-4ade-97b0-10743764a2c7","resolution":{"observed_at":"2026-08-05T11:01:22.134368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.100472Z","title":"Sardi, R","venue":null,"work_id":"8390718c-27c4-429e-a048-f265b3223a89","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.689452Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:6ef729b3e01511bf52c2c9ee942ee638e8010d35c529838a3ef67d20afd915aa","observation_id":"7f54e2f7-6208-4e42-9c1c-9c0756e75334","resolution":{"observed_at":"2026-08-05T11:01:22.111796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.078290Z","title":"Sardi, R","venue":null,"work_id":"d1b1996f-f10c-4939-85a3-6a0262639342","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.702248Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:ef55b51333aa15b35901eed5451433db5a2534d341fff6fcc275c9baf9000f41","observation_id":"31ee5467-8ef9-482b-a2c0-25afc4242c3a","resolution":{"observed_at":"2026-08-05T11:01:22.084562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.053851Z","title":"Lehmann, R","venue":null,"work_id":"439f8b15-0b2d-4e9b-9302-1bab41a00d29","year":2015},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.709874Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:ad18fb886fa578b711eae4f505fd236c837e23567008978fcc32d142faa4ea8f","observation_id":"88b366c5-2bc6-4520-8773-64d63095f5a1","resolution":{"observed_at":"2026-08-05T11:01:22.059447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.10147","last_updated":"2018-10-26T21:41:46Z","snapshot_observed_at":"2026-07-06T07:10:12.570041Z","submitted_at":"2018-10-24T01:18:08Z","title":"FewRel: A Large-Scale Supervised Few-Shot Relation Classification Dataset with State-of-the-Art Evaluation","version":2},"cited_work":{"arxiv_id":"1810.10147","doi":null,"metadata_source":"pith","pith_arxiv_id":"1810.10147","snapshot_observed_at":"2026-08-05T11:01:21.019258Z","title":"FewRel: A Large-Scale Supervised Few-Shot Relation Classification Dataset with State-of-the-Art Evaluation","venue":"cs.LG","work_id":"6ed74376-479a-4245-a5be-c11127413078","year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.716385Z"},"links":{"cited_paper":"/paper/1810.10147","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:e844dcf27496b50d6b6210a9a8fd27d46b4c6400e66796163078d8842bc7fa78","observation_id":"897b0e9f-4ec2-42d3-ad53-81c119e8f2a5","resolution":{"observed_at":"2026-08-05T11:01:21.031563Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.01703","last_updated":"2019-12-03T22:06:05Z","snapshot_observed_at":"2026-07-06T08:41:49.632205Z","submitted_at":"2019-12-03T22:06:05Z","title":"PyTorch: An Imperative Style, High-Performance Deep Learning Library","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.01703","snapshot_observed_at":"2026-08-05T11:01:20.721924Z","title":"Paszke, Pytorch: An imperative style, high-performance deep learning library, arXiv preprint arXiv:1912.01703, (2019)","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.721924Z"},"links":{"cited_paper":"/paper/1912.01703","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:e7b6da2b856c042f3a97c22c35265fea6535f3ddacedd7b4db53933a12bb63b2","observation_id":"0b75307d-e708-46d6-a36f-b77cd6988d53","resolution":{"observed_at":"2026-08-05T11:01:20.721924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:22.023791Z","title":"Schmidhuber, Deep learning in neural networks: An overview, Neural networks, 61 (2015) 85-117","venue":null,"work_id":"d51c1ad2-95f3-4098-adab-1ca618e3497a","year":2015},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.730997Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:225f91fdd9907b629db9f79e981472f50dbc6b405a46e6ca39db8baaf34a8488","observation_id":"56758630-999d-49dc-9607-00868fa711b7","resolution":{"observed_at":"2026-08-05T11:01:22.034442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:01:21.987879Z","title":null,"venue":null,"work_id":"94e03c72-ad69-4f9a-8e68-d9f998977759","year":2016},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.740269Z"},"links":{"citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:1daf11fbc01557fe9c0964b983426e73921c3d6c871afa26446c87503e7c0325","observation_id":"243c5f76-3ab2-4309-92f5-6e3fd730eb53","resolution":{"observed_at":"2026-08-05T11:01:21.995577Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T11:01:20.747158Z","title":"Loshchilov, F","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.747158Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:6a47941e2a4aad5e21b20eac577a88382a6741eebcba92b672b5fb6c9f1436f3","observation_id":"d1921562-adaf-4b16-be88-8dcc9c8b2458","resolution":{"observed_at":"2026-08-05T11:01:20.747158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1205.2653","last_updated":"2012-05-09T15:01:22Z","snapshot_observed_at":"2026-08-10T16:00:09.318909Z","submitted_at":"2012-05-09T15:01:22Z","title":"L2 Regularization for Learning Kernels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1205.2653","snapshot_observed_at":"2026-08-05T11:01:20.753605Z","title":"Cortes, M","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.753605Z"},"links":{"cited_paper":"/paper/1205.2653","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:d104417288e18b60ec9e59605bb0d08212068967e019ca44a6cb95e2d6cfac3e","observation_id":"5e8d4bdd-1d62-48c6-a26b-b800afd71024","resolution":{"observed_at":"2026-08-05T11:01:20.753605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.02677","last_updated":"2018-04-30T21:53:41Z","snapshot_observed_at":"2026-08-09T05:23:26.365677Z","submitted_at":"2017-06-08T16:51:53Z","title":"Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.02677","snapshot_observed_at":"2026-08-05T11:01:20.761518Z","title":"Goyal, P","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.761518Z"},"links":{"cited_paper":"/paper/1706.02677","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:2d476fe040e3611e5143a25c994da8fdf941d355b3de3c6f67152fb1aeec3d53","observation_id":"1616d138-a0ff-49aa-8b08-130635d6a4c1","resolution":{"observed_at":"2026-08-05T11:01:20.761518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01108","last_updated":"2020-03-01T02:57:50Z","snapshot_observed_at":"2026-08-07T19:07:36.327251Z","submitted_at":"2019-10-02T17:56:28Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01108","snapshot_observed_at":"2026-08-05T11:01:20.769972Z","title":null,"venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.769972Z"},"links":{"cited_paper":"/paper/1910.01108","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:6b355bb812544966d2ee96b62869d6b81a651e07ca16575a85897e6c8701f116","observation_id":"6a970119-8bd2-4196-84cc-644e09281535","resolution":{"observed_at":"2026-08-05T11:01:20.769972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1801.06146","last_updated":"2018-05-23T09:23:47Z","snapshot_observed_at":"2026-08-06T18:01:48.915710Z","submitted_at":"2018-01-18T17:54:52Z","title":"Universal Language Model Fine-tuning for Text Classification","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1801.06146","snapshot_observed_at":"2026-08-05T11:01:20.775942Z","title":"Howard, S","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.775942Z"},"links":{"cited_paper":"/paper/1801.06146","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:ee814d0f0a03252a2ea830782398874ce2501deeb0dec7bed98573ea7c4cdc56","observation_id":"57488ebc-df3e-4b84-b68b-d210824b2946","resolution":{"observed_at":"2026-08-05T11:01:20.775942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":7,"verified_fuzzy":33},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 0 inbound Pith citation observations for arXiv:2509.03407."}