{"as_of":"2026-08-21T03:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e366dd5b79c837b283a2ad62a2536a901e4d618c26804a3d9473424343c5be13","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T16:57:23.947375Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.13056/citation-record","integrity":"/paper/2411.13056/integrity","json":"/paper/2411.13056/citation-record.json","paper":"/paper/2411.13056"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.782447Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.782447Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:966f86da6980ce05a6ef76d403189c05e46098bb7601ac2509391a2f26d5e752","observation_id":"78de6edd-6ba5-4416-a0f8-b5a9de3591b8","resolution":{"observed_at":"2026-08-12T16:57:23.782447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:25.049189Z","title":"Arteta, V","venue":null,"work_id":"32c3e52c-9c0b-4fb4-bbea-3a23cab14a0e","year":2016},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.787353Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:d8636fe9bea33f819b5c88250a735970753568aa6c9747cfa06de240555b4393","observation_id":"c2614720-692f-476a-8d3f-41d5899fd900","resolution":{"observed_at":"2026-08-12T16:57:25.052902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.791186Z","title":"A spatio-temporal attentive network for video-based crowd counting","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.791186Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:8543b5a91ffa396b78df55eda03aabe4577f3720cb1073cdab75832f13da7cad","observation_id":"9336d576-683b-4a9e-a2a2-e54644fc83db","resolution":{"observed_at":"2026-08-12T16:57:23.791186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:25.039139Z","title":"MultiMAE : Multi-modal multi-task masked autoencoders","venue":null,"work_id":"9da4a691-6383-42e5-909e-1fc96d753996","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.795090Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:58a9168afa672d88e8cea5d27106fd8d8e8dd1808fd555bce6c3982164489879","observation_id":"52a8ef27-0ddc-4792-b827-281bb1618396","resolution":{"observed_at":"2026-08-12T16:57:25.042712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:25.029123Z","title":null,"venue":null,"work_id":"a7140a3f-8c6d-4870-a31e-2d3adf9cbf04","year":2021},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.798655Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:d39c425223478bdab1546a5934a864d5b15d2f893e254e423fce81384bde98a0","observation_id":"78f9d0d9-06a9-4221-9460-8fd303c44804","resolution":{"observed_at":"2026-08-12T16:57:25.032456Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.802121Z","title":"BE it: BERT pre-training of image transformers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.802121Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:6f5e41f275e134ce2c290a32b7045395dee9fdfc69cda6f18052aa58b186500e","observation_id":"b1d05056-85ce-473d-ae0e-94583a8d7f67","resolution":{"observed_at":"2026-08-12T16:57:23.802121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:25.013297Z","title":"Generative pretraining from pixels","venue":null,"work_id":"7aa33f7c-63b3-4c0b-bd08-796fe3115071","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.805406Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:590bfc52e5d5b85bb6bb32a993bf48d5ca3c4f4124c35af33813085f085d9fb0","observation_id":"ab34073d-7ad5-4ec8-85b8-cc4932a56fcc","resolution":{"observed_at":"2026-08-12T16:57:25.016749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.809064Z","title":"BERT : Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.809064Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:51b6492e8a428e2c7a6f37a5393edc2f83a78aea66c570aa22637607ecff8906","observation_id":"7559e6d6-2689-4933-babd-80b49dffba1c","resolution":{"observed_at":"2026-08-12T16:57:23.809064Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T16:57:23.812413Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.812413Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:ab081eacea689ce72b8450be126cfa0dad3c6525f73dd7008d5c1d4428f678e1","observation_id":"440ed128-4068-4d88-aa34-4a7999eff820","resolution":{"observed_at":"2026-08-12T16:57:23.812413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:25.003257Z","title":"Redesigning multi-scale neural network for crowd counting","venue":null,"work_id":"fba6770d-f97c-4c14-ad47-ab7ce0677e42","year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.815966Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:273670b2b73e62220cb551c9774ac21759764f08c5da52495e737328a3c6e9e5","observation_id":"9685b215-d4aa-48d5-a5d6-15a924bd2266","resolution":{"observed_at":"2026-08-12T16:57:25.007105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.07911","last_updated":"2019-07-18T07:25:26Z","snapshot_observed_at":"2026-08-19T13:29:56.479033Z","submitted_at":"2019-07-18T07:25:26Z","title":"Locality-constrained Spatial Transformer Network for Video Crowd Counting","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.07911","snapshot_observed_at":"2026-08-12T16:57:23.819308Z","title":"Locality-constrained spatial transformer network for video crowd counting","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.819308Z"},"links":{"cited_paper":"/paper/1907.07911","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:e7c9e4971bd15c49c36ee85fad38b13a4d40a281019b2da87bf69826c5690082","observation_id":"626e37d2-a91a-4636-9903-0addb52abe50","resolution":{"observed_at":"2026-08-12T16:57:23.819308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.neucom.2020.01.087","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.015300Z","title":"Multi-level feature fusion based locality-constrained spatial transformer network for video crowd counting","venue":null,"work_id":"07386a9f-df13-4cdc-9348-fab8ec31cbb5","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.823305Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:d7b93d78e2d16d31b41c24ba1afe94e746956c53df5933b114d30bd225bd97a6","observation_id":"371c6409-ac7b-4fe8-9441-4d53923a1442","resolution":{"observed_at":"2026-08-12T16:57:24.018963Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.06377","last_updated":"2021-12-19T19:23:25Z","snapshot_observed_at":"2026-08-17T17:58:26.288570Z","submitted_at":"2021-11-11T18:46:40Z","title":"Masked Autoencoders Are Scalable Vision Learners","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.06377","snapshot_observed_at":"2026-08-12T16:57:23.826976Z","title":"Masked autoencoders are scalable vision learners","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.826976Z"},"links":{"cited_paper":"/paper/2111.06377","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:567de428a056854d3a1216da1f292004bb3f0e7d00dbbba542b8ac42f9d4608e","observation_id":"d4e7bb18-c7e9-4865-be09-3b03414112da","resolution":{"observed_at":"2026-08-12T16:57:23.826976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.992344Z","title":"Video-based crowd counting using a multi-scale optical flow pyramid network","venue":null,"work_id":"dd839afd-29eb-43dc-9e23-b9acc799531c","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.830574Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:72a5cd57632b499e8a8c23e60ab02079397271c64b2856ff78e3d186be7459e5","observation_id":"dd01dfb4-1de7-4279-a4ee-3027a3802743","resolution":{"observed_at":"2026-08-12T16:57:24.996512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.833699Z","title":"Frame-recurrent video crowd counting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.833699Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:1c2efbe06c636995e2ba2ac79a9166135863099db666ac040402e8fd1460f7fa","observation_id":"9db75f98-6b03-47ea-a154-dc0b13e0e245","resolution":{"observed_at":"2026-08-12T16:57:23.833699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.837157Z","title":"Clip-count: Towards text-guided zero-shot object counting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.837157Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:e744790ebe7a7881b8d4406b414d1f4c9c4c18651aa4e05f2c795f61292c1627","observation_id":"d57eee47-f21d-4ffc-81d4-41e1418f6ed8","resolution":{"observed_at":"2026-08-12T16:57:23.837157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.982025Z","title":"Vlcounter: Text-aware visual representation for zero-shot object counting","venue":null,"work_id":"d4d0aea0-eda7-4a3b-a37a-de29e189141e","year":2024},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.840443Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:d76dd9fff5d22c2ea93fa0a3582c588b75393a8db66535d10c3f557338d9eeb3","observation_id":"b07f420c-f39f-4c4b-bbdc-2760dec31ff1","resolution":{"observed_at":"2026-08-12T16:57:24.985644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2022.32052","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.601023Z","title":"Video crowd localization with multifocus gaussian neighborhood attention and a large-scale benchmark","venue":null,"work_id":"fbbe9635-cbd5-412c-be43-d2c8670c568b","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.843597Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:fbc31619fa6225b259b0a742a5194ccff03529c7e8312c6c8c53d7b4a07565ad","observation_id":"44abdf7a-f553-4568-9b82-d0e3f0096172","resolution":{"observed_at":"2026-08-12T16:57:24.609996Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.846966Z","title":"Csrnet: Dilated convolutional neural networks for understanding the highly congested scenes","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.846966Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:798c9971f8db6c5c0d99266b91988300f73bf604e16a6988ce657f771545e6cb","observation_id":"43b5dbcc-1851-4bbe-b4aa-145e236db61d","resolution":{"observed_at":"2026-08-12T16:57:23.846966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.972044Z","title":"Transcrowd: weakly-supervised crowd counting with transformers","venue":null,"work_id":"88340db1-faf7-4281-8eb8-69a5245c1252","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.853888Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:364cad3a1eb97515773c5478bf54f95d439b880d8a66f878276e1aa7e735e05e","observation_id":"7e7bb7f6-0872-48bc-9c94-cd381b828bad","resolution":{"observed_at":"2026-08-12T16:57:24.975681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.961724Z","title":"Boosting crowd counting via multifaceted attention","venue":null,"work_id":"78193fb6-756e-4bd4-bfc3-31f7114f19c6","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.857463Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:0c6620f8aab74260e3109d376249d3be43df8f94225219d5ec721b37f429b0b7","observation_id":"f8569ae5-1d0d-48be-a570-e427f6b725be","resolution":{"observed_at":"2026-08-12T16:57:24.965341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03870","last_updated":"2024-01-08T13:01:54Z","snapshot_observed_at":"2026-08-16T14:29:08.369612Z","submitted_at":"2024-01-08T13:01:54Z","title":"Gramformer: Learning Crowd Counting via Graph-Modulated Transformer","version":1},"cited_work":{"arxiv_id":"2401.03870","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.03870","snapshot_observed_at":"2026-08-12T16:57:24.469153Z","title":"Gramformer: Learning Crowd Counting via Graph-Modulated Transformer","venue":"cs.CV","work_id":"12977b4c-b8d6-41af-8cab-7b9d3ba0d24d","year":2024},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.860634Z"},"links":{"cited_paper":"/paper/2401.03870","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:7b21aa83eff588551889e6d6318567746b9025b33a2a3ae2206a5eb93fe36ee4","observation_id":"cd0653db-3a7d-43e6-9cdd-0a97e685dace","resolution":{"observed_at":"2026-08-12T16:57:24.474037Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.951669Z","title":"Point-query quadtree for crowd counting, localization, and more","venue":null,"work_id":"0d53795d-c806-4494-8ff3-6f7405ad739c","year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.863958Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:6509d95432ca6b068a92cc8abfc786e31b628ab0e93d5a87604e3ae60a21a0b4","observation_id":"d159f02c-9c61-4f5d-b4a5-f0a345eddb72","resolution":{"observed_at":"2026-08-12T16:57:24.955272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.941577Z","title":"Context-aware crowd counting","venue":null,"work_id":"67c49a30-4a3f-4cb1-b9f9-bdf4d89aa02e","year":2019},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.867202Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:47d0e09ecc7d6d238fe9baf3a6602aef6d4c2f9ad5235d8ad8c11b70455b3f05","observation_id":"7a0b16c9-afd0-4c71-9896-e1652db2a2ee","resolution":{"observed_at":"2026-08-12T16:57:24.945253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.931152Z","title":"Estimating people flows to better count them in crowded scenes","venue":null,"work_id":"a78c1d8a-868f-4534-82a9-374c83cd8291","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.870428Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:eaa911b0f5b45d08aaab12dbca49da0e8cdb7e8f99c527158a9da6260009e315","observation_id":"664e97d6-0f59-46a2-990b-63d94dcb3050","resolution":{"observed_at":"2026-08-12T16:57:24.934936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1109/iccv.2013.270","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.005420Z","title":"From semi-supervised to transfer counting of crowds","venue":null,"work_id":"e6d79240-2985-478d-8f19-421fafdfa4c8","year":2013},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.873558Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:db088ed75a3e894f530fa9bd1e16d0297fa7149236360d4ff68ed9bcbcb3fdf2","observation_id":"aa9eff38-6cd3-4e1a-aec1-b52ec02577b8","resolution":{"observed_at":"2026-08-12T16:57:24.008980Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.920540Z","title":"Bayesian loss for crowd count estimation with point supervision","venue":null,"work_id":"a72390ce-9299-425d-a6da-176a7b0ea698","year":2019},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.876933Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:9b745ffd3286994dba9cb5fcf75e087f95f6e2b7e466fd63e3522b786d7d4660","observation_id":"166a18b5-fc40-4574-9de1-af9f9e8641b0","resolution":{"observed_at":"2026-08-12T16:57:24.924155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.880503Z","title":"Phnet: Parasite-host network for video crowd counting","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.880503Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:46fbc540d629cb7ed2a6231d64e28ca51c223f287a08bc78d7ca3446f69c8943","observation_id":"ecad169a-ae56-4f8c-bd91-ec3133ddc848","resolution":{"observed_at":"2026-08-12T16:57:23.880503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.883550Z","title":"Rudin, Stanley Osher, and Emad Fatemi","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.883550Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:4a8afea1b14996baa02f4b7e1de83ba764a34cfb79ac0d25cf01389ee2ebd198","observation_id":"e3544976-f0bd-4017-a5b4-3356690211c3","resolution":{"observed_at":"2026-08-12T16:57:23.883550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.909894Z","title":"Convolutional lstm network: a machine learning approach for precipitation nowcasting","venue":null,"work_id":"198111d4-8239-4a21-8344-923b9f4067fb","year":2015},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.886796Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:25b174ec56c121814567c2f45792c016df56d8e60fd83853bf00fbbef60f850f","observation_id":"5bb89ac6-e73a-40be-8d23-01b8fa385da9","resolution":{"observed_at":"2026-08-12T16:57:24.913583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.899828Z","title":"Crowd counting in the frequency domain","venue":null,"work_id":"ab01c676-cdb0-443b-aaae-2bc0e846989f","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.889842Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:92b31660d16689f66e7b4ab3b1cd6bec62591379f6f678d4f46b5e3a3451b6ca","observation_id":"91c7fe4f-fb45-4a2d-be96-89aa511175b8","resolution":{"observed_at":"2026-08-12T16:57:24.903338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.889320Z","title":"PWC-Net : CNNs for optical flow using pyramid, warping, and cost volume","venue":null,"work_id":"4a10cdc3-bb22-4f4d-a414-b4308c567d7c","year":2018},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.893078Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:2ed1d1dcfa963db0c32d9e25e40d73e98e2256b6ddafab7f1e795f2e7d614280","observation_id":"42c4479e-a51b-402e-9b9b-5cd1d2c1b31a","resolution":{"observed_at":"2026-08-12T16:57:24.893026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14483","last_updated":"2021-09-29T15:13:10Z","snapshot_observed_at":"2026-08-20T10:03:37.678442Z","submitted_at":"2021-09-29T15:13:10Z","title":"CCTrans: Simplifying and Improving Crowd Counting with Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14483","snapshot_observed_at":"2026-08-12T16:57:23.896271Z","title":"Cctrans: Simplifying and improving crowd counting with transformer","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.896271Z"},"links":{"cited_paper":"/paper/2109.14483","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:7e362e39ec0a8a661e82c421f805f6bd110a22d8343dbcefcaa737c99e8fe4fc","observation_id":"d51332c6-d2ce-48cf-a815-15be8225e664","resolution":{"observed_at":"2026-08-12T16:57:23.896271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.878700Z","title":"Video MAE : Masked autoencoders are data-efficient learners for self-supervised video pre-training","venue":null,"work_id":"b905a7fd-a298-4aaf-85db-1ceecc2d8e46","year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.899774Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:c991d5cfbf00a3e2e41714f590899cb7b49435a9bc7b48dbeb5d993153a8989f","observation_id":"3e5a2c63-1336-4c00-9265-d6ff21e9ba3b","resolution":{"observed_at":"2026-08-12T16:57:24.882248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.903168Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.903168Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:77fd20e88015625b006f15e31e45fd7208ccc6944df87e94b8ea63f044774db9","observation_id":"8c05006d-6a9c-4f66-a70d-d29040192035","resolution":{"observed_at":"2026-08-12T16:57:23.903168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.906726Z","title":"Extracting and composing robust features with denoising autoencoders","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.906726Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:f1bbd0aa695ec2bbb56e28c17f52f935ddceee665d013391c98325e386f3a70a","observation_id":"0b6566b4-245c-4135-9a4b-6423cc651fa1","resolution":{"observed_at":"2026-08-12T16:57:23.906726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11042-023-14833-z","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-19T11:03:45.297681Z","title":"Bird-count: a multi-modality benchmark and system for bird population counting in the wild","venue":"Multimedia Tools and Applications","work_id":"68e1bb18-5f3f-48fa-86d6-5731dd6be98b","year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.910115Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:60398570decebc0cbec5e3187f6d821610060bcc9405d0ec71e1ddb22e441c20","observation_id":"c431bf60-4e15-4b29-accb-0a01d9669159","resolution":{"observed_at":"2026-08-12T16:57:23.998950Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.862116Z","title":"Fast video crowd counting with a temporal aware network","venue":null,"work_id":"ef7be2ec-85cc-4279-aaa9-ed1d70f366cb","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.913415Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:6cd9ad7216fac87561ebeec12bd1fa7179b6acf85947e586ad7d5b0735b1c723","observation_id":"a2bd0bf0-a4b8-4aa7-a698-fce5016d0455","resolution":{"observed_at":"2026-08-12T16:57:24.866031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.916569Z","title":"Spatial-temporal graph network for video crowd counting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.916569Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:2cb63d440f800c118e57fbf71120b27ba73b903188a6abb20ce8404188ed4422","observation_id":"c3b0ec67-e2ae-441f-b134-c66cba37238b","resolution":{"observed_at":"2026-08-12T16:57:23.916569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1109/iccv.2017.551","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.980070Z","title":"Spatiotemporal modeling for crowd counting in videos","venue":null,"work_id":"62df393c-4cb8-4cfa-8e0a-8a3e8106c0cf","year":2017},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.919832Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:e368d718f480f2d84d9acffee23576203e2d5f554f2de78db0feb3e7bf0f554b","observation_id":"8c06c77d-2b8a-43bc-a1d8-7d2e5070bc3a","resolution":{"observed_at":"2026-08-12T16:57:23.985587Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.851527Z","title":"Reverse perspective network for perspective-aware object counting","venue":null,"work_id":"2e05df30-2b70-4aae-8b5a-4038e9728f56","year":2020},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.923135Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:6554ac391252fb47d0a8595202fe9640d75dd94c7c0a430a4007481d65944322","observation_id":"470da1b4-0df2-4d25-9f9c-e3fa5f45fd4f","resolution":{"observed_at":"2026-08-12T16:57:24.855306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.840960Z","title":"Single-image crowd counting via multi-column convolutional neural network","venue":null,"work_id":"db741f4d-9f64-4e04-b6d6-8e327fa6bb5b","year":2016},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.926246Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:7914c06d1019fecbf89479f71da3a4bb451d109247d05196a19f293453495bd1","observation_id":"7a3fbda3-32c5-486a-9fdc-48b6ff5a7707","resolution":{"observed_at":"2026-08-12T16:57:24.844758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.929315Z","title":"Locality-aware crowd counting","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.929315Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:a60e93e47b8a400256faa000c9542f49fdc4bd2d5a672d4f325ffdf306cb33d4","observation_id":"c75526b9-4f58-4993-b322-03008a703bc4","resolution":{"observed_at":"2026-08-12T16:57:23.929315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2021.30822","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:24.129551Z","title":"Graph regularized flow attention network for video animal counting from drones","venue":null,"work_id":"172ca934-b2f0-45a2-bfd7-3e2fb8f3a2ac","year":2021},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.932906Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:9388b2c9d51cfdebc042d0f0f07257498217ce2e70bec28ab7e73e8a4c7f9a81","observation_id":"b7e0ffea-fd1d-4f8a-bf2b-0649ab05a780","resolution":{"observed_at":"2026-08-12T16:57:24.135086Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.04121","last_updated":"2019-08-12T12:39:28Z","snapshot_observed_at":"2026-08-16T14:20:07.963052Z","submitted_at":"2019-08-12T12:39:28Z","title":"Enhanced 3D convolutional networks for crowd counting","version":1},"cited_work":{"arxiv_id":"1908.04121","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.04121","snapshot_observed_at":"2026-08-12T16:57:24.035685Z","title":"Enhanced 3D convolutional networks for crowd counting","venue":"cs.CV","work_id":"3e824681-c14b-4894-98b5-ed2ff69cdbee","year":2019},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.936167Z"},"links":{"cited_paper":"/paper/1908.04121","citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:b130385176f2773c00e15e22f921d038a15228ce6e4003502f2ac1393031ce1f","observation_id":"f90697e8-e4f4-4dab-9989-1d5d94f0adc5","resolution":{"observed_at":"2026-08-12T16:57:24.039410Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.939779Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.939779Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:9c3c628513ca21f3942b2c5708a29a27ad5935a5df238d53d68fc6de13a728bb","observation_id":"7d9bc860-b41b-481b-bf69-a368c78472d8","resolution":{"observed_at":"2026-08-12T16:57:23.939779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.943688Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.943688Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:8ae0ab91ee809f2f46d2cc4d7bdc1f1e434a47a7fd2c5e79bd8ca663de0ec13f","observation_id":"2c6a71cc-5246-406f-9fea-608bfe245980","resolution":{"observed_at":"2026-08-12T16:57:23.943688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:57:23.947375Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-12T16:57:23.947375Z"},"links":{"citing_paper":"/paper/2411.13056"},"observation_digest":"sha256:6da3aa2d29a2c480e6e0e6058ae74db657f4e1435140bfb31d59371a56435095","observation_id":"0fe7fb19-91b2-479c-842f-0ee17427ea7c","resolution":{"observed_at":"2026-08-12T16:57:23.947375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.13056","last_updated":"2025-03-06T08:28:09Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T13:53:57.555378Z","submitted_at":"2024-11-20T06:08:21Z","title":"Efficient Masked AutoEncoder for Video Object Counting and A Large-Scale Benchmark"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":20,"verified_exact":6,"verified_fuzzy":19},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 0 inbound Pith citation observations for arXiv:2411.13056."}