{"as_of":"2026-08-13T16:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4266a37aafd83f68380fb29317ed1a3147027a8d1043e5806fc4692728b5fa2c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":6,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":6,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T11:19:33.862479Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-05T11:41:02.799655Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":"2106.04803","doi":null,"metadata_source":"pith","pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-07-05T11:41:02.799655Z","title":"Coatnet: Marrying convolution and attention for all data sizes","venue":"cs.CV","work_id":"08de913e-bf2d-4981-af29-094eb833d77d","year":2021},"citing_paper":{"arxiv_id":"2110.02178","last_updated":"2022-03-04T17:17:31Z","snapshot_observed_at":"2026-07-06T11:54:38.041673Z","submitted_at":"2021-10-05T17:07:53Z","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T20:46:35.073600Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2110.02178"},"observation_digest":"sha256:2b2254ad265123f72eb8c01266b9d7b6b6954ac271be7ce50288315759fb5f34","observation_id":"1e08dccc-3a69-42d3-b1e7-7942c5e5b7ba","resolution":{"observed_at":"2026-05-20T20:46:35.211861Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":"2106.04803","doi":null,"metadata_source":"pith","pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-07-05T11:41:02.799655Z","title":"Coatnet: Marrying convolution and attention for all data sizes","venue":"cs.CV","work_id":"08de913e-bf2d-4981-af29-094eb833d77d","year":2021},"citing_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T09:38:09.427509Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2111.11432"},"observation_digest":"sha256:d7018f1af0b7b6d5cdc0b522872580f69dae9b0df4a89e09ee217809bdb41979","observation_id":"bd6b2b45-d640-41e6-972f-b6f4603c7867","resolution":{"observed_at":"2026-05-16T09:38:09.485888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-08-10T11:19:33.862479Z","title":"Coatnet: Marrying convolution and attention for all data sizes,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16704","last_updated":"2025-01-28T04:46:50Z","snapshot_observed_at":"2026-08-10T13:00:40.658094Z","submitted_at":"2025-01-28T04:46:50Z","title":"DFCon: Attention-Driven Supervised Contrastive Learning for Robust Deepfake Detection","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T11:19:33.862479Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2501.16704"},"observation_digest":"sha256:af5ddab4c87461371075611e82bafb6ed272b988fb363d8e1487063391c50429","observation_id":"07410519-75ec-4f48-99b6-0727dd33450b","resolution":{"observed_at":"2026-08-10T11:19:33.862479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-08-06T23:21:45.974828Z","title":"V., & Tan, M","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.18683","last_updated":"2025-06-23T14:25:40Z","snapshot_observed_at":"2026-08-08T19:58:42.308580Z","submitted_at":"2025-06-23T14:25:40Z","title":"SIM-Net: A Multimodal Fusion Network Using Inferred 3D Object Shape Point Clouds from RGB Images for 2D Classification","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:21:45.974828Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2506.18683"},"observation_digest":"sha256:3ade6d6c26898d76505f1160c20834bc2b846764b7801fd6d12d2e0879a6b924","observation_id":"2c9c4fae-ac83-4d4e-8b8e-4d89912c42c4","resolution":{"observed_at":"2026-08-06T23:21:45.974828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":"2106.04803","doi":null,"metadata_source":"pith","pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-07-05T11:41:02.799655Z","title":"Coatnet: Marrying convolution and attention for all data sizes","venue":"cs.CV","work_id":"08de913e-bf2d-4981-af29-094eb833d77d","year":2021},"citing_paper":{"arxiv_id":"2604.18543","last_updated":"2026-06-10T02:43:26Z","snapshot_observed_at":"2026-08-12T18:47:26.431104Z","submitted_at":"2026-04-20T17:36:49Z","title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-05T11:32:36.356530Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2604.18543"},"observation_digest":"sha256:819a644e57fe8da2be70f570162fa600131abd1681bbf6dd0aa7c6456ecf9972","observation_id":"3c15361b-e1a1-4a4e-bef6-28156de9a622","resolution":{"observed_at":"2026-07-05T11:41:02.801223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes","version":2},"cited_work":{"arxiv_id":"2106.04803","doi":null,"metadata_source":"pith","pith_arxiv_id":"2106.04803","snapshot_observed_at":"2026-07-05T11:41:02.799655Z","title":"Coatnet: Marrying convolution and attention for all data sizes","venue":"cs.CV","work_id":"08de913e-bf2d-4981-af29-094eb833d77d","year":2021},"citing_paper":{"arxiv_id":"2604.18549","last_updated":"2026-04-20T17:41:00Z","snapshot_observed_at":"2026-08-03T01:01:56.885161Z","submitted_at":"2026-04-20T17:41:00Z","title":"Advancing Vision Transformer with Enhanced Spatial Priors","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T05:22:21.264807Z"},"links":{"cited_paper":"/paper/2106.04803","citing_paper":"/paper/2604.18549"},"observation_digest":"sha256:1dac638e2b058f73ace104ab8f88054bb7b2abdba14a0fa207fd909f11c59671","observation_id":"30f27239-2882-49e5-bec0-8f7fd7a4ed2c","resolution":{"observed_at":"2026-05-10T09:23:37.664602Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2106.04803/citation-record","integrity":"/paper/2106.04803/integrity","json":"/paper/2106.04803/citation-record.json","paper":"/paper/2106.04803"},"outbound":[],"paper":{"arxiv_id":"2106.04803","last_updated":"2021-09-15T06:05:13Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T07:16:34.641640Z","submitted_at":"2021-06-09T04:35:31Z","title":"CoAtNet: Marrying Convolution and Attention for All Data Sizes"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 6 inbound Pith citation observations for arXiv:2106.04803."}